rulereceipt 0.1.35 → 0.1.37

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -88,6 +88,7 @@ rulereceipt rules --include <handle> # "this IS a rule" — check it from now
88
88
  rulereceipt rules --exclude <handle> # "this isn't" — stop reporting it
89
89
  rulereceipt rules --coverage # which rules a configured hook might actually enforce
90
90
  rulereceipt doctor # list hooks/auto-run tasks configured on this machine
91
+ rulereceipt hook # run AS a Claude Code Stop hook — block Claude finishing on a broken rule
91
92
  rulereceipt lint # find contradictions between CLAUDE.md and AGENTS.md
92
93
  rulereceipt digest # summarise recent checks; --email to send it
93
94
  rulereceipt config # set up email sending (stays on your machine)
@@ -99,6 +100,49 @@ rulereceipt verify <session-file> <hash> # spot-check a report you received ag
99
100
 
100
101
  `verify` isn't a routine check — trust your team day to day, same as any status update. It's there for the rare case it actually matters (a dispute, an incident review): give it the session file and the hash printed in the report, and it confirms whether they really match.
101
102
 
103
+ ## Blocking, not just reporting
104
+
105
+ `rulereceipt check` tells you afterwards. `rulereceipt hook` refuses to let the
106
+ session end.
107
+
108
+ Add this to `.claude/settings.json` — you add it, we never do:
109
+
110
+ ```json
111
+ {
112
+ "hooks": {
113
+ "Stop": [
114
+ { "hooks": [ { "type": "command", "command": "npx rulereceipt hook" } ] }
115
+ ]
116
+ }
117
+ }
118
+ ```
119
+
120
+ When Claude tries to finish, it reads the session that just happened. If a rule
121
+ was broken it hands Claude the rule, the evidence, and instructions to keep
122
+ working, so the session cannot end on a claim that isn't backed.
123
+
124
+ The one it is actually for: *"Done — all tests pass"* when the last run of
125
+ `npm test` returned two failures. It checks the claim against what ran, which
126
+ is the part a model cannot talk its way around.
127
+
128
+ Three properties worth knowing before you wire it in:
129
+
130
+ - **It only blocks on things it can prove.** Never a judgment rule, never an
131
+ LLM opinion, never "couldn't tell". Only a matched literal or a claim
132
+ contradicted by a recorded tool result. Run against twelve real sessions it
133
+ blocked none of them.
134
+ - **It cannot loop.** Claude Code sets `stop_hook_active` when a session is
135
+ already continuing because of a block; the hook returns immediately in that
136
+ case. One interruption per stop.
137
+ - **It fails open.** Unreadable transcript, missing rules file, a bug in us —
138
+ it allows the stop and writes a line to stderr. Failing closed would mean our
139
+ bug locks you out of finishing your own session. That is a deliberate
140
+ weakening, and it is why `check` in CI stays the backstop.
141
+
142
+ It runs when Claude stops, so it catches a finished session, not a command
143
+ mid-flight. For that, use a `PreToolUse` hook of your own — `rulereceipt
144
+ doctor` will show you what you already have.
145
+
102
146
  ## Which rules actually have teeth
103
147
 
104
148
  A rule in a file and a rule with a `PreToolUse` hook behind it look identical
@@ -1,4 +1,4 @@
1
- import { TEST_COMMAND } from "./testCommands.js";
1
+ import { TEST_COMMAND, withoutHeredocs, countTestRuns } from "./testCommands.js";
2
2
  /**
3
3
  * Did the session claim something worked, when the log says it didn't?
4
4
  *
@@ -200,7 +200,7 @@ export function runClaimEvidenceChecks(classifications, events) {
200
200
  for (const event of events) {
201
201
  const command = commandOf(event);
202
202
  if (command !== null) {
203
- pendingRun = TEST_COMMAND.test(command) ? command : null;
203
+ pendingRun = TEST_COMMAND.test(withoutHeredocs(command)) ? command : null;
204
204
  for (const action of ACTION_CLAIMS) {
205
205
  if (action.command.test(command))
206
206
  commandsSeen.add(action.label);
@@ -222,10 +222,15 @@ export function runClaimEvidenceChecks(classifications, events) {
222
222
  // words survive a pipe, the exit status does not.
223
223
  const stated = outcomeFromOutput(event.content);
224
224
  const trustExitCode = !PIPED.test(pendingRun);
225
+ // Two suite invocations in one command means neither the exit code
226
+ // nor the printed summary belongs to a single run, so nothing about
227
+ // this command can contradict a claim. Checked before both, because
228
+ // reading the output is what defeated the pipe guard here.
229
+ const oneRun = countTestRuns(pendingRun) <= 1;
225
230
  lastRun = {
226
231
  command: pendingRun,
227
232
  failed: stated !== null ? stated : event.isError,
228
- outcomeReadable: stated !== null || trustExitCode,
233
+ outcomeReadable: oneRun && (stated !== null || trustExitCode),
229
234
  output: event.content.slice(0, 200),
230
235
  };
231
236
  pendingRun = null;
@@ -200,19 +200,83 @@ const PRE_ACTION_GATE = /\b(?:repeat[- ]back|restate\s+what|wait\s+for\s+(?:conf
200
200
  * failures on its own.
201
201
  */
202
202
  const MAX_RULE_BODY_FOR_CLAIM = 1500;
203
+ /**
204
+ * The claim shape: a reporting verb NEXT TO a done-word.
205
+ *
206
+ * "Do not claim tests passed", "before declaring work complete", "before
207
+ * telling user publishing is complete" — the two halves sit within a few
208
+ * words of each other, in one sentence, because that adjacency IS the
209
+ * thing being forbidden: announcing a finish you have not verified.
210
+ *
211
+ * Testing the two words independently across a whole section is what kept
212
+ * mis-routing. A 944-character rule about writing Slack updates contained a
213
+ * reporting verb in one paragraph, an evidence noun in another, and the
214
+ * done-word inside the compound "Slack-ready" — three unrelated matches,
215
+ * one confident verdict about whether a test command was piped. That was
216
+ * patched twice with blocklist entries (PRE_ACTION_GATE) before the
217
+ * independence was recognised as the fault itself.
218
+ *
219
+ * Thirty characters, measured rather than picked: it is the smallest
220
+ * window that keeps every genuine claim rule in the corpus and the widest
221
+ * that admits none of the scattered ones. `[^.\n]` confines the match to a
222
+ * single sentence, which is what stops a match spanning a paragraph break.
223
+ * Across 559 public rules files this narrows claim-evidence from 38 rules
224
+ * to 8, and all 8 read as the same instruction in different words.
225
+ */
226
+ const CLAIM_SHAPE = new RegExp(`(?:${REPORTING_VERB.source})[^.\n]{0,30}?(?:${DONE_WORD.source})` +
227
+ `|(?:${DONE_WORD.source})[^.\n]{0,30}?(?:${REPORTING_VERB.source})`, "i");
203
228
  function isClaimEvidenceRule(rule) {
204
229
  if (rule.text.length > MAX_RULE_BODY_FOR_CLAIM)
205
230
  return false;
206
231
  const text = `${rule.title} ${rule.text}`;
232
+ // Kept as a second line of defence even though CLAIM_SHAPE now excludes
233
+ // every case it was added for. It is cheap, it is tested, and the rules
234
+ // it names are ones this checker must never answer.
207
235
  if (PRE_ACTION_GATE.test(text))
208
236
  return false;
209
- return REPORTING_VERB.test(text) && EVIDENCE_NOUN.test(text) && DONE_WORD.test(text);
237
+ return CLAIM_SHAPE.test(text) && EVIDENCE_NOUN.test(text);
210
238
  }
211
239
  const BRANCH_WORD = /\bbranch\b/i;
240
+ /**
241
+ * A literal that could actually be a git branch name.
242
+ *
243
+ * git's refname rules, reduced to what a CLAUDE.md really writes: no
244
+ * whitespace, no shell punctuation, no `..`, no leading dash, not ending
245
+ * `.lock`. Everything this excludes was being accepted as a branch name and
246
+ * then searched for in the session's git commands.
247
+ *
248
+ * Rejecting a template is the point of the character class rather than an
249
+ * accident of it: `{`, `<` and `$` are how a rules file writes a PATTERN
250
+ * for branch names ("squad/{issue-number}-{kebab-case-slug}"), and a
251
+ * pattern is not a branch. Checking whether the session used a branch
252
+ * literally called that has no meaningful answer.
253
+ */
254
+ const BRANCH_NAME_PATTERN = /^[A-Za-z0-9._/-]{1,100}$/;
255
+ function isBranchName(literal) {
256
+ if (!BRANCH_NAME_PATTERN.test(literal))
257
+ return false;
258
+ if (literal.startsWith("-") || literal.startsWith("/"))
259
+ return false;
260
+ if (literal.includes("..") || literal.endsWith(".lock") || literal.endsWith("/"))
261
+ return false;
262
+ return true;
263
+ }
212
264
  // A function/method-call shape ("print(", "analytics.track(") is a strong,
213
265
  // simple signal that a backtick literal names actual CODE, not a CLI
214
266
  // command or flag ("git push --force", "npm test" never look like this).
215
- const CODE_CONSTRUCT_PATTERN = /\(/;
267
+ //
268
+ // It must be a CALL, not merely a parenthesis. The test used to be /\(/,
269
+ // which is true of a great deal of ordinary prose: measured 2026-09-15
270
+ // across 559 rules files, 74 of 1,086 literals reaching content matching
271
+ // (6.8%) were not code — "(e.g.", "(soft)", a markdown link fragment, three
272
+ // whole blocks of accounting formulae. "(in the" produced a real false
273
+ // accusation, reported as having been "actually written into a file", which
274
+ // is true of any file containing that phrase.
275
+ //
276
+ // An identifier immediately before the paren, optionally dotted or scoped,
277
+ // so "console.log(", "std::cout(" and "obj->run(" all qualify and a bare
278
+ // parenthesis does not.
279
+ const CODE_CONSTRUCT_PATTERN = /[A-Za-z_$][A-Za-z0-9_$]*(?:\s*(?:\.|::|->)\s*[A-Za-z_$][A-Za-z0-9_$]*)*\s*\(/;
216
280
  // A file-path shape: a known config/source extension, or a path with a
217
281
  // directory separator. Deliberately requires no spaces — a real path
218
282
  // literal ("`.claude/settings.json`", "`config.yaml`") never has one,
@@ -441,11 +505,12 @@ export function classifyRule(rule) {
441
505
  return { kind: "judgment", rule };
442
506
  }
443
507
  const polarityInferred = polarityWasInferred(rule);
444
- if (BRANCH_WORD.test(text)) {
445
- // first backtick literal is treated as the branch name — real rules
446
- // this targets name exactly one branch ("the `demo` branch", "never
447
- // push to `main`"), not a set of them
448
- const [branchName] = patterns;
508
+ // The first literal that could BE a branch, not simply the first literal.
509
+ // Real rules here name exactly one branch ("the `demo` branch", "never
510
+ // push to the `main` branch"), and a rule that mentions branches while
511
+ // naming a command belongs to whichever checker handles that command.
512
+ const branchName = [...patterns].find(isBranchName);
513
+ if (BRANCH_WORD.test(text) && branchName !== undefined) {
449
514
  return { kind: "gitBranchPolicy", rule, branchName, polarity, polarityInferred };
450
515
  }
451
516
  // Route on ANY code-shaped literal, but check ONLY the code-shaped ones.
@@ -29,6 +29,37 @@ function editedContentFromEvent(event) {
29
29
  return input.new_source;
30
30
  return null;
31
31
  }
32
+ /**
33
+ * Whether the content contains this literal AS A CALL, not merely as a
34
+ * substring of a longer identifier.
35
+ *
36
+ * Found 2026-09-15 by checking a corpus FAIL rather than assuming it was
37
+ * legitimate: a rule forbidding `fetch()` matched a file containing
38
+ * `_metar_fetch()`. The literal was present verbatim, and entirely the wrong
39
+ * function. The same bare-substring test makes `main()` match `domain()` and
40
+ * `run()` match `rerun()`, and short generic call names are exactly what
41
+ * these rules tend to name.
42
+ *
43
+ * Only the LEADING boundary is checked. The trailing side is already pinned
44
+ * by the pattern itself — every literal reaching this checker ends in an
45
+ * open paren or a call — so requiring a boundary after it would reject the
46
+ * arguments.
47
+ */
48
+ function containsCall(content, pattern) {
49
+ const leadsWithIdentifier = /^[A-Za-z0-9_$]/.test(pattern);
50
+ if (!leadsWithIdentifier)
51
+ return content.includes(pattern);
52
+ let from = 0;
53
+ for (;;) {
54
+ const at = content.indexOf(pattern, from);
55
+ if (at === -1)
56
+ return false;
57
+ const before = at === 0 ? "" : content[at - 1];
58
+ if (!/[A-Za-z0-9_$.]/.test(before))
59
+ return true;
60
+ from = at + 1;
61
+ }
62
+ }
32
63
  export function runCodeContentChecks(classifications, events) {
33
64
  const editedContents = [];
34
65
  for (const event of events) {
@@ -41,7 +72,7 @@ export function runCodeContentChecks(classifications, events) {
41
72
  let foundContent;
42
73
  for (const content of editedContents) {
43
74
  for (const pattern of patterns) {
44
- if (content.includes(pattern)) {
75
+ if (containsCall(content, pattern)) {
45
76
  foundPattern = pattern;
46
77
  foundContent = content;
47
78
  break;
@@ -1,5 +1,6 @@
1
1
  import { violation } from "../types.js";
2
2
  import { isProjectPath } from "./projectPaths.js";
3
+ import { withoutHeredocs } from "./shellCommand.js";
3
4
  /**
4
5
  * Third structured-check primitive: only counts real MUTATIONS of a
5
6
  * protected file, never reads of it. Real false-positive this fixes
@@ -43,7 +44,17 @@ function pathPattern(filePath) {
43
44
  * it mutates after that is a scratch file, not the project's.
44
45
  */
45
46
  const CD_INTO_TEMP = /\bcd\s+["']?(?:\/private)?\/(?:tmp|var\/folders)\b|\bcd\s+["']?[^\s"'&|;]*\/(?:scratchpad|node_modules)\b/;
46
- function mutatesPathInBash(command, filePath) {
47
+ /**
48
+ * A path named only inside a heredoc body was not touched by the command
49
+ * that contains it.
50
+ *
51
+ * Real case, 2026-09-15: a command editing landing/index.html through a
52
+ * Python heredoc was reported as modifying `.claude/`, because the HTML it
53
+ * inserts tells readers to put a hook in `.claude/settings.json`. Writing a
54
+ * path into a file is not mutating that path.
55
+ */
56
+ function mutatesPathInBash(rawCommand, filePath) {
57
+ const command = withoutHeredocs(rawCommand);
47
58
  const p = pathPattern(filePath);
48
59
  const mutations = [
49
60
  // rm / rmdir / unlink targeting the path
@@ -0,0 +1,24 @@
1
+ /**
2
+ * Facts about shell command text that more than one checker needs.
3
+ *
4
+ * Created 2026-09-15. The heredoc guard below lived in testCommands.ts, was
5
+ * used only by the test-command matcher, and the file-mutation checker never
6
+ * saw it — so the same class of false accusation was fixed in one reader and
7
+ * left standing in the other. Anything that reasons about what a shell
8
+ * command DID, rather than what it says, belongs here.
9
+ */
10
+ /**
11
+ * Removes heredoc bodies from a shell command.
12
+ *
13
+ * A command that WRITES a test command is not a command that RUNS one.
14
+ * Found 2026-09-14 on a real session: two false failures whose "last test
15
+ * run" was a shell variable assignment. The actual match came from a
16
+ * heredoc further down, writing a demo fixture whose body contains the
17
+ * string `npm test`. The literal was being generated, never executed — and
18
+ * the tool then read its own report output as the failing result.
19
+ *
20
+ * Handles both quoted and bare delimiters, and leaves everything after the
21
+ * closing delimiter intact, because a real test run often follows the
22
+ * heredoc that set the fixture up.
23
+ */
24
+ export declare function withoutHeredocs(command: string): string;
@@ -0,0 +1,43 @@
1
+ /**
2
+ * Facts about shell command text that more than one checker needs.
3
+ *
4
+ * Created 2026-09-15. The heredoc guard below lived in testCommands.ts, was
5
+ * used only by the test-command matcher, and the file-mutation checker never
6
+ * saw it — so the same class of false accusation was fixed in one reader and
7
+ * left standing in the other. Anything that reasons about what a shell
8
+ * command DID, rather than what it says, belongs here.
9
+ */
10
+ /**
11
+ * Removes heredoc bodies from a shell command.
12
+ *
13
+ * A command that WRITES a test command is not a command that RUNS one.
14
+ * Found 2026-09-14 on a real session: two false failures whose "last test
15
+ * run" was a shell variable assignment. The actual match came from a
16
+ * heredoc further down, writing a demo fixture whose body contains the
17
+ * string `npm test`. The literal was being generated, never executed — and
18
+ * the tool then read its own report output as the failing result.
19
+ *
20
+ * Handles both quoted and bare delimiters, and leaves everything after the
21
+ * closing delimiter intact, because a real test run often follows the
22
+ * heredoc that set the fixture up.
23
+ */
24
+ export function withoutHeredocs(command) {
25
+ const lines = command.split("\n");
26
+ const out = [];
27
+ let closing = null;
28
+ for (const line of lines) {
29
+ if (closing !== null) {
30
+ if (line.trim() === closing)
31
+ closing = null;
32
+ continue;
33
+ }
34
+ const open = line.match(/<<-?\s*(?:'([^']+)'|"([^"]+)"|([A-Za-z_][A-Za-z0-9_]*))/);
35
+ if (open) {
36
+ closing = open[1] ?? open[2] ?? open[3];
37
+ out.push(line.slice(0, open.index));
38
+ continue;
39
+ }
40
+ out.push(line);
41
+ }
42
+ return out.join("\n");
43
+ }
@@ -19,5 +19,24 @@ import type { TranscriptEvent } from "../types.js";
19
19
  * wrong.
20
20
  */
21
21
  export declare const TEST_COMMAND: RegExp;
22
+ /**
23
+ * How many times a single shell command invokes a test suite.
24
+ *
25
+ * A command that runs the suite twice has no single outcome to attribute.
26
+ * Real case, 2026-09-14: one command ran a typecheck, then the suite (37
27
+ * passed), then re-ran one file with locking deliberately disabled to
28
+ * demonstrate those tests can fail. The session said "Everything passes:",
29
+ * which was true; the checker read "2 failed" out of the third section and
30
+ * called it a lie.
31
+ *
32
+ * Deliberate red runs cannot be recognised - "is this failure intended" is
33
+ * not in the text. What IS in the text is that the suite ran more than
34
+ * once, and that is enough: the outcome is unattributable for the same
35
+ * reason a pipe makes an exit code unattributable. Heredoc bodies are
36
+ * stripped first, so a command that merely writes a test command twice
37
+ * still counts zero.
38
+ */
39
+ export declare function countTestRuns(command: string): number;
22
40
  /** The first test command run in this session, or null if none ran. */
23
41
  export declare function findTestRun(events: TranscriptEvent[]): string | null;
42
+ export { withoutHeredocs } from "./shellCommand.js";
@@ -1,3 +1,4 @@
1
+ import { withoutHeredocs } from "./shellCommand.js";
1
2
  /**
2
3
  * Commands that run a project's test suite.
3
4
  *
@@ -18,6 +19,27 @@
18
19
  * wrong.
19
20
  */
20
21
  export const TEST_COMMAND = /\b(?:npm|pnpm|yarn|bun)\s+(?:run\s+)?(?:test|verify|check|ci)\b|\bnpx\s+(?:vitest|jest|mocha|ava)\b|\b(?:vitest|jest|mocha|pytest|phpunit|rspec|tox)\b|\bcargo\s+test\b|\bgo\s+test\b|\bmvn\s+(?:test|verify)\b|\bgradle\s+test\b|\bdotnet\s+test\b|\bpython\s+-m\s+(?:pytest|unittest)\b/i;
22
+ /**
23
+ * How many times a single shell command invokes a test suite.
24
+ *
25
+ * A command that runs the suite twice has no single outcome to attribute.
26
+ * Real case, 2026-09-14: one command ran a typecheck, then the suite (37
27
+ * passed), then re-ran one file with locking deliberately disabled to
28
+ * demonstrate those tests can fail. The session said "Everything passes:",
29
+ * which was true; the checker read "2 failed" out of the third section and
30
+ * called it a lie.
31
+ *
32
+ * Deliberate red runs cannot be recognised - "is this failure intended" is
33
+ * not in the text. What IS in the text is that the suite ran more than
34
+ * once, and that is enough: the outcome is unattributable for the same
35
+ * reason a pipe makes an exit code unattributable. Heredoc bodies are
36
+ * stripped first, so a command that merely writes a test command twice
37
+ * still counts zero.
38
+ */
39
+ export function countTestRuns(command) {
40
+ const global = new RegExp(TEST_COMMAND.source, "gi");
41
+ return (withoutHeredocs(command).match(global) ?? []).length;
42
+ }
21
43
  /** The first test command run in this session, or null if none ran. */
22
44
  export function findTestRun(events) {
23
45
  for (const event of events) {
@@ -25,8 +47,9 @@ export function findTestRun(events) {
25
47
  continue;
26
48
  const input = event.input;
27
49
  const command = input && typeof input.command === "string" ? input.command : "";
28
- if (command && TEST_COMMAND.test(command))
50
+ if (command && TEST_COMMAND.test(withoutHeredocs(command)))
29
51
  return command;
30
52
  }
31
53
  return null;
32
54
  }
55
+ export { withoutHeredocs } from "./shellCommand.js";
package/dist/cli.js CHANGED
@@ -16,6 +16,7 @@ import { runGitBranchPolicyChecks } from "./checks/gitBranchPolicy.js";
16
16
  import { runCodeContentChecks } from "./checks/codeContent.js";
17
17
  import { runFileLifecycleChecks } from "./checks/fileLifecycle.js";
18
18
  import { runClaimEvidenceChecks } from "./checks/claimEvidence.js";
19
+ import { runHook } from "./hook.js";
19
20
  import { runJudgmentChecks } from "./checks/judgmentChecks.js";
20
21
  import { generateReport, generateMarkdownReport } from "./report/generateReport.js";
21
22
  import { generateHtmlReport } from "./report/generateHtmlReport.js";
@@ -573,6 +574,12 @@ function runDoctorCommand() {
573
574
  console.log(`${result.newSinceLastRun.length} of these are new since the last time doctor ran here.`);
574
575
  }
575
576
  }
577
+ program
578
+ .command("hook")
579
+ .description("run as a Claude Code Stop hook - blocks Claude from finishing on a broken rule (payload on stdin)")
580
+ .action(async () => {
581
+ await runHook(needsLlmResult);
582
+ });
576
583
  program
577
584
  .command("doctor")
578
585
  .description("List every Claude Code hook and VS Code auto-task on this machine/project, flag anything suspicious")
@@ -0,0 +1,22 @@
1
+ import { type Classification } from "./checks/classify.js";
2
+ import { staleOverrides } from "./overrides.js";
3
+ import type { CheckResult, Rule, TranscriptEvent } from "./types.js";
4
+ export interface Evaluation {
5
+ results: CheckResult[];
6
+ notARule: Classification[];
7
+ stale: ReturnType<typeof staleOverrides>;
8
+ }
9
+ /**
10
+ * Rules in, verdicts out — the whole pipeline, with no printing in it.
11
+ *
12
+ * Extracted 2026-09-14 when a second caller appeared. `check` renders a
13
+ * report for a person; `hook` returns a decision to Claude Code. Those two
14
+ * must never be able to disagree about whether a rule was broken, and the
15
+ * only way to guarantee that is one body of code. The same reasoning is
16
+ * written at the top of testCommands.ts, about two checkers sharing one
17
+ * definition; this is that argument one level up.
18
+ *
19
+ * Deliberately takes `events` rather than a path: the caller decides where
20
+ * a transcript comes from, and a hook is handed one it must not second-guess.
21
+ */
22
+ export declare function evaluateSession(cwd: string, rules: Rule[], events: TranscriptEvent[], llm: boolean, needsLlmResult: (rule: Rule) => CheckResult): Promise<Evaluation>;
@@ -0,0 +1,51 @@
1
+ import { classifyRules } from "./checks/classify.js";
2
+ import { runDeterministicChecks } from "./checks/deterministicChecks.js";
3
+ import { runIfEditThenTestChecks } from "./checks/ifEditThenTest.js";
4
+ import { runGitBranchPolicyChecks } from "./checks/gitBranchPolicy.js";
5
+ import { runCodeContentChecks } from "./checks/codeContent.js";
6
+ import { runFileLifecycleChecks } from "./checks/fileLifecycle.js";
7
+ import { runClaimEvidenceChecks } from "./checks/claimEvidence.js";
8
+ import { runJudgmentChecks } from "./checks/judgmentChecks.js";
9
+ import { loadOverrides, ruleFingerprint, staleOverrides } from "./overrides.js";
10
+ /**
11
+ * Rules in, verdicts out — the whole pipeline, with no printing in it.
12
+ *
13
+ * Extracted 2026-09-14 when a second caller appeared. `check` renders a
14
+ * report for a person; `hook` returns a decision to Claude Code. Those two
15
+ * must never be able to disagree about whether a rule was broken, and the
16
+ * only way to guarantee that is one body of code. The same reasoning is
17
+ * written at the top of testCommands.ts, about two checkers sharing one
18
+ * definition; this is that argument one level up.
19
+ *
20
+ * Deliberately takes `events` rather than a path: the caller decides where
21
+ * a transcript comes from, and a hook is handed one it must not second-guess.
22
+ */
23
+ export async function evaluateSession(cwd, rules, events, llm, needsLlmResult) {
24
+ const overrides = loadOverrides(cwd);
25
+ const classifications = classifyRules(rules).map((c) => {
26
+ const decision = overrides.get(ruleFingerprint(c.rule))?.decision;
27
+ if (!decision)
28
+ return c;
29
+ if (decision === "notARule")
30
+ return { kind: "notARule", rule: c.rule };
31
+ return c.kind === "notARule" ? { kind: "judgment", rule: c.rule } : c;
32
+ });
33
+ const of = (kind) => classifications.filter((c) => c.kind === kind);
34
+ const deterministicResults = [
35
+ ...runDeterministicChecks(of("deterministic"), events),
36
+ ...runIfEditThenTestChecks(of("ifEditThenTest"), events),
37
+ ...runGitBranchPolicyChecks(of("gitBranchPolicy"), events),
38
+ ...runCodeContentChecks(of("codeContent"), events),
39
+ ...runFileLifecycleChecks(of("fileLifecycle"), events),
40
+ ...runClaimEvidenceChecks(of("claimEvidence"), events),
41
+ ];
42
+ const judgment = of("judgment");
43
+ const judgmentResults = llm
44
+ ? await runJudgmentChecks(judgment, events)
45
+ : judgment.map(({ rule }) => needsLlmResult(rule));
46
+ return {
47
+ results: [...deterministicResults, ...judgmentResults],
48
+ notARule: of("notARule"),
49
+ stale: staleOverrides(overrides, rules),
50
+ };
51
+ }
package/dist/hook.d.ts ADDED
@@ -0,0 +1,35 @@
1
+ import type { CheckResult, Rule } from "./types.js";
2
+ /**
3
+ * A Stop hook that refuses to let a session end on a broken rule.
4
+ *
5
+ * The gap this closes was named by a reader of anthropics/claude-code#90542
6
+ * on 2026-09-14, from a lab that measured it: a receiving agent auto-ACKed
7
+ * 107 dispatched orders and did none of them, and 69 of 664 records marked
8
+ * complete were plans rather than completions. Their conclusion — "rules
9
+ * that live only in context are advisory by construction; rules that live
10
+ * in a gate are not" — applies to this tool as it stood, which read the
11
+ * transcript afterwards and told you what had already happened.
12
+ *
13
+ * Three safety properties, in the order they matter:
14
+ *
15
+ * 1. FAIL only. Never UNCLEAR, never NOT_RUN, never a judgment rule, and
16
+ * never the LLM path — a model opinion is an opinion, and blocking on
17
+ * one would let a wrong guess hold a session hostage. This gate fires
18
+ * only on the deterministic checkers, where the finding is a matched
19
+ * literal and can be read back.
20
+ *
21
+ * 2. It cannot loop. Claude Code sets `stop_hook_active` when the session
22
+ * is already continuing because of a previous block. Blocking again
23
+ * there is how a Stop hook wedges a session permanently, so this returns
24
+ * immediately in that case: it gets exactly one interruption per stop.
25
+ *
26
+ * 3. It fails OPEN. Any error — unreadable transcript, no rules file,
27
+ * malformed payload — allows the stop and writes a line to stderr.
28
+ * Failing closed is the right default for money-touching code; here it
29
+ * would mean a bug in this tool leaves someone unable to end a session
30
+ * in their own editor, and they would rip the hook out that day. The
31
+ * cost of the two failures is not symmetric, so the default is not
32
+ * symmetric either. It is a deliberate weakening, and it is the reason
33
+ * `check` in CI stays the backstop rather than this.
34
+ */
35
+ export declare function runHook(needsLlmResult: (rule: Rule) => CheckResult): Promise<void>;
package/dist/hook.js ADDED
@@ -0,0 +1,100 @@
1
+ import { loadRules } from "./rules.js";
2
+ import { readTranscriptFromFile } from "./parsers/transcriptParser.js";
3
+ import { evaluateSession } from "./evaluate.js";
4
+ function readStdin() {
5
+ return new Promise((resolve) => {
6
+ let data = "";
7
+ if (process.stdin.isTTY)
8
+ return resolve("");
9
+ process.stdin.setEncoding("utf8");
10
+ process.stdin.on("data", (c) => (data += c));
11
+ process.stdin.on("end", () => resolve(data));
12
+ process.stdin.on("error", () => resolve(""));
13
+ });
14
+ }
15
+ /**
16
+ * What Claude is told when it tries to finish on a broken rule.
17
+ *
18
+ * Addressed to the model, not to the person: it is the model that has to
19
+ * act on this, and a message written for a human reader ("see the report
20
+ * above") gives it nothing to do. Names the rule, states what was found,
21
+ * and stops — no instruction to "fix it", because the rule already said
22
+ * what to do and repeating it invites the model to argue with the wording
23
+ * instead of going and running the thing.
24
+ */
25
+ function blockReason(failures) {
26
+ const lines = failures.map((f) => ` • Rule ${f.ruleId} — ${f.ruleTitle}\n ${f.evidence}`);
27
+ const n = failures.length;
28
+ return (`RuleReceipt: ${n} rule${n === 1 ? "" : "s"} in CLAUDE.md ${n === 1 ? "was" : "were"} not followed in this session.\n\n` +
29
+ lines.join("\n\n") +
30
+ `\n\nDo not report this work as finished until the above is resolved or you have said plainly, to the user, that it is still open and why.`);
31
+ }
32
+ /**
33
+ * A Stop hook that refuses to let a session end on a broken rule.
34
+ *
35
+ * The gap this closes was named by a reader of anthropics/claude-code#90542
36
+ * on 2026-09-14, from a lab that measured it: a receiving agent auto-ACKed
37
+ * 107 dispatched orders and did none of them, and 69 of 664 records marked
38
+ * complete were plans rather than completions. Their conclusion — "rules
39
+ * that live only in context are advisory by construction; rules that live
40
+ * in a gate are not" — applies to this tool as it stood, which read the
41
+ * transcript afterwards and told you what had already happened.
42
+ *
43
+ * Three safety properties, in the order they matter:
44
+ *
45
+ * 1. FAIL only. Never UNCLEAR, never NOT_RUN, never a judgment rule, and
46
+ * never the LLM path — a model opinion is an opinion, and blocking on
47
+ * one would let a wrong guess hold a session hostage. This gate fires
48
+ * only on the deterministic checkers, where the finding is a matched
49
+ * literal and can be read back.
50
+ *
51
+ * 2. It cannot loop. Claude Code sets `stop_hook_active` when the session
52
+ * is already continuing because of a previous block. Blocking again
53
+ * there is how a Stop hook wedges a session permanently, so this returns
54
+ * immediately in that case: it gets exactly one interruption per stop.
55
+ *
56
+ * 3. It fails OPEN. Any error — unreadable transcript, no rules file,
57
+ * malformed payload — allows the stop and writes a line to stderr.
58
+ * Failing closed is the right default for money-touching code; here it
59
+ * would mean a bug in this tool leaves someone unable to end a session
60
+ * in their own editor, and they would rip the hook out that day. The
61
+ * cost of the two failures is not symmetric, so the default is not
62
+ * symmetric either. It is a deliberate weakening, and it is the reason
63
+ * `check` in CI stays the backstop rather than this.
64
+ */
65
+ export async function runHook(needsLlmResult) {
66
+ const emit = (out) => {
67
+ process.stdout.write(JSON.stringify(out));
68
+ };
69
+ try {
70
+ const raw = await readStdin();
71
+ const input = raw ? JSON.parse(raw) : {};
72
+ // Property 2: one interruption per stop.
73
+ if (input.stop_hook_active)
74
+ return void emit({});
75
+ const cwd = input.cwd || process.cwd();
76
+ if (!input.transcript_path)
77
+ return void emit({});
78
+ const rules = loadRules(cwd);
79
+ if (rules.length === 0)
80
+ return void emit({});
81
+ const events = readTranscriptFromFile(input.transcript_path);
82
+ // An empty transcript produces no failures, which would read as a pass.
83
+ // Nothing to gate on, so allow — and say nothing, because a hook that
84
+ // warns on every empty read is a hook people mute.
85
+ if (events.length === 0)
86
+ return void emit({});
87
+ // Property 1: deterministic only. `llm: false` is not a default here,
88
+ // it is part of the contract.
89
+ const { results } = await evaluateSession(cwd, rules, events, false, needsLlmResult);
90
+ const failures = results.filter((r) => r.status === "FAIL" && r.outcome !== "not_run");
91
+ if (failures.length === 0)
92
+ return void emit({});
93
+ return void emit({ decision: "block", reason: blockReason(failures) });
94
+ }
95
+ catch (err) {
96
+ // Property 3: fail open, but never silently.
97
+ process.stderr.write(`rulereceipt hook: allowing stop, check did not complete (${err instanceof Error ? err.message : String(err)})\n`);
98
+ return void emit({});
99
+ }
100
+ }
@@ -154,10 +154,27 @@ const BUCKET_ORDER = [
154
154
  function sharedEvidence(rs) {
155
155
  if (rs.length < 2)
156
156
  return null;
157
- const first = rs[0].evidence;
158
- if (!first)
159
- return null;
160
- return rs.every((r) => r.evidence === first) ? first : null;
157
+ // The MAJORITY text, not a unanimous one. Requiring every entry to match
158
+ // meant a single rule with its own message reinstated the wall for all the
159
+ // others — shipped in 0.1.35 and visible at once: fourteen judgment rules,
160
+ // twelve repeating the same 300-character paragraph, because two carried
161
+ // claim-evidence text. Whether it happened depended on what was in the
162
+ // transcript that minute, so it passed locally and broke for the reader.
163
+ const counts = new Map();
164
+ for (const r of rs) {
165
+ if (!r.evidence)
166
+ continue;
167
+ counts.set(r.evidence, (counts.get(r.evidence) ?? 0) + 1);
168
+ }
169
+ let best = null;
170
+ let bestN = 1;
171
+ for (const [text, n] of counts) {
172
+ if (n > bestN) {
173
+ best = text;
174
+ bestN = n;
175
+ }
176
+ }
177
+ return best;
161
178
  }
162
179
  export function generateReport(results, meta) {
163
180
  const clean = results.map(sanitize);
@@ -177,8 +194,15 @@ export function generateReport(results, meta) {
177
194
  if (shared) {
178
195
  lines.push(` ${shared}`);
179
196
  lines.push("");
180
- for (const r of inBucket)
197
+ for (const r of inBucket) {
181
198
  lines.push(` ${ruleLabel(r, clean)}`);
199
+ // An entry that does not share the hoisted text still says its own
200
+ // piece — that difference is the only per-rule information there is.
201
+ if (r.evidence && r.evidence !== shared)
202
+ lines.push(` ${r.evidence}`);
203
+ if (r.ceiling)
204
+ lines.push(` this means: ${r.ceiling}`);
205
+ }
182
206
  continue;
183
207
  }
184
208
  for (const r of inBucket) {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "rulereceipt",
3
- "version": "0.1.35",
3
+ "version": "0.1.37",
4
4
  "description": "Checks whether a Claude Code session actually followed your CLAUDE.md / AGENTS.md rules, with evidence.",
5
5
  "repository": {
6
6
  "type": "git",