rulereceipt 0.1.36 → 0.1.38

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -89,6 +89,7 @@ rulereceipt rules --exclude <handle> # "this isn't" — stop reporting it
89
89
  rulereceipt rules --coverage # which rules a configured hook might actually enforce
90
90
  rulereceipt doctor # list hooks/auto-run tasks configured on this machine
91
91
  rulereceipt hook # run AS a Claude Code Stop hook — block Claude finishing on a broken rule
92
+ rulereceipt guard # run AS a Claude Code PreToolUse hook — refuse a call before it runs
92
93
  rulereceipt lint # find contradictions between CLAUDE.md and AGENTS.md
93
94
  rulereceipt digest # summarise recent checks; --email to send it
94
95
  rulereceipt config # set up email sending (stays on your machine)
@@ -127,10 +128,16 @@ is the part a model cannot talk its way around.
127
128
 
128
129
  Three properties worth knowing before you wire it in:
129
130
 
130
- - **It only blocks on things it can prove.** Never a judgment rule, never an
131
- LLM opinion, never "couldn't tell". Only a matched literal or a claim
132
- contradicted by a recorded tool result. Run against twelve real sessions it
133
- blocked none of them.
131
+ - **It blocks two things, both narrow.** A claim a recorded run contradicts,
132
+ and a claim of done that nothing in the session verified. Never a judgment
133
+ rule, never an LLM opinion. Run against thirteen real sessions it stopped
134
+ two, and both were read by hand.
135
+ - **The report and the gate disagree in exactly one place.** When a session
136
+ claims work is done and nothing recorded verifies it, the report says
137
+ "couldn't tell" — the tests may have run in another terminal, and a
138
+ transcript cannot see that. The gate refuses the exit anyway, because it is
139
+ not saying the claim is false. It is declining to let "done" end a session
140
+ with nothing behind it.
134
141
  - **It cannot loop.** Claude Code sets `stop_hook_active` when a session is
135
142
  already continuing because of a block; the hook returns immediately in that
136
143
  case. One interruption per stop.
@@ -143,6 +150,36 @@ It runs when Claude stops, so it catches a finished session, not a command
143
150
  mid-flight. For that, use a `PreToolUse` hook of your own — `rulereceipt
144
151
  doctor` will show you what you already have.
145
152
 
153
+ ### Refusing a command before it runs
154
+
155
+ `rulereceipt guard` runs as a `PreToolUse` hook and refuses a call outright:
156
+
157
+ ```json
158
+ {
159
+ "hooks": {
160
+ "PreToolUse": [
161
+ { "hooks": [ { "type": "command", "command": "npx rulereceipt guard" } ] }
162
+ ]
163
+ }
164
+ }
165
+ ```
166
+
167
+ Read the limit before wiring it in, because it is most of the story. It
168
+ enforces rules naming a **file** or a **branch** — "never modify `.env`",
169
+ "never commit to `main`" — and nothing else.
170
+
171
+ It does **not** block banned commands. That was the point of building it, and
172
+ it did not survive measurement: replaying 16,336 real tool calls against every
173
+ forbidding rule in a 559-file corpus, blocking on command literals refused
174
+ 62.8% of them. Narrowing twice reached 2.5%, and the residue was still wrong
175
+ in a way no matcher fixes — one rule refused `npm run build` 112 times,
176
+ because it forbids running Playwright unprompted and *recommends*
177
+ `npm run build`, which is its only command-shaped literal.
178
+
179
+ Nothing in a rules file marks which backtick is the prohibition. A report
180
+ survives that by saying UNCLEAR. A gate cannot.
181
+
182
+
146
183
  ## Which rules actually have teeth
147
184
 
148
185
  A rule in a file and a rule with a `PreToolUse` hook behind it look identical
@@ -188,7 +188,11 @@ function unclear(rule, evidence) {
188
188
  */
189
189
  export function runClaimEvidenceChecks(classifications, events) {
190
190
  let lastRun = null;
191
- let pendingRun = null;
191
+ // Keyed by tool_use id where the transcript has one, so a result can be
192
+ // matched to the call it belongs to rather than to the call above it.
193
+ // `null` is the key for id-less transcripts, which keeps the old
194
+ // positional behaviour for fixtures and older logs.
195
+ const pendingRuns = new Map();
192
196
  let claimsMade = 0;
193
197
  let unknownSinceRed = null;
194
198
  const commandsSeen = new Set();
@@ -200,7 +204,15 @@ export function runClaimEvidenceChecks(classifications, events) {
200
204
  for (const event of events) {
201
205
  const command = commandOf(event);
202
206
  if (command !== null) {
203
- pendingRun = TEST_COMMAND.test(withoutHeredocs(command)) ? command : null;
207
+ const id = event.kind === "tool_use" ? (event.toolUseId ?? null) : null;
208
+ if (TEST_COMMAND.test(withoutHeredocs(command))) {
209
+ pendingRuns.set(id, command);
210
+ }
211
+ else if (id === null) {
212
+ // No id to distinguish calls, so a later call really does supersede
213
+ // an earlier one — the original positional rule, unchanged.
214
+ pendingRuns.delete(null);
215
+ }
204
216
  for (const action of ACTION_CLAIMS) {
205
217
  if (action.command.test(command))
206
218
  commandsSeen.add(action.label);
@@ -217,6 +229,8 @@ export function runClaimEvidenceChecks(classifications, events) {
217
229
  // every turn contained exactly one tool call, so there is no parallel
218
230
  // fan-out to mis-attribute.
219
231
  if (event.kind === "tool_result") {
232
+ const resultId = event.toolUseId ?? null;
233
+ const pendingRun = pendingRuns.get(resultId) ?? null;
220
234
  if (pendingRun !== null) {
221
235
  // Prefer what the runner SAID over what the shell returned: the
222
236
  // words survive a pipe, the exit status does not.
@@ -233,7 +247,7 @@ export function runClaimEvidenceChecks(classifications, events) {
233
247
  outcomeReadable: oneRun && (stated !== null || trustExitCode),
234
248
  output: event.content.slice(0, 200),
235
249
  };
236
- pendingRun = null;
250
+ pendingRuns.delete(resultId);
237
251
  unknownSinceRed = null; // a recognised run supersedes anything before it
238
252
  }
239
253
  continue;
@@ -331,8 +345,12 @@ export function runClaimEvidenceChecks(classifications, events) {
331
345
  };
332
346
  }
333
347
  if (claimsMade > 0) {
334
- return unclear(rule, `the session claimed a passing test suite ${claimsMade} time(s), but no test command ran here — ` +
335
- `it may have been run outside this session, which the transcript cannot show`);
348
+ return {
349
+ ...unclear(rule, `the session claimed a passing test suite ${claimsMade} time(s), but no test command ran here — ` +
350
+ `it may have been run outside this session, which the transcript cannot show`),
351
+ // UNCLEAR in the report, refused by the gate. See CheckResult.unverifiedClaim.
352
+ unverifiedClaim: true,
353
+ };
336
354
  }
337
355
  return unclear(rule, "the session made no claim about passing tests, so there was nothing to check against the log");
338
356
  });
@@ -264,7 +264,19 @@ function isBranchName(literal) {
264
264
  // A function/method-call shape ("print(", "analytics.track(") is a strong,
265
265
  // simple signal that a backtick literal names actual CODE, not a CLI
266
266
  // command or flag ("git push --force", "npm test" never look like this).
267
- const CODE_CONSTRUCT_PATTERN = /\(/;
267
+ //
268
+ // It must be a CALL, not merely a parenthesis. The test used to be /\(/,
269
+ // which is true of a great deal of ordinary prose: measured 2026-09-15
270
+ // across 559 rules files, 74 of 1,086 literals reaching content matching
271
+ // (6.8%) were not code — "(e.g.", "(soft)", a markdown link fragment, three
272
+ // whole blocks of accounting formulae. "(in the" produced a real false
273
+ // accusation, reported as having been "actually written into a file", which
274
+ // is true of any file containing that phrase.
275
+ //
276
+ // An identifier immediately before the paren, optionally dotted or scoped,
277
+ // so "console.log(", "std::cout(" and "obj->run(" all qualify and a bare
278
+ // parenthesis does not.
279
+ const CODE_CONSTRUCT_PATTERN = /[A-Za-z_$][A-Za-z0-9_$]*(?:\s*(?:\.|::|->)\s*[A-Za-z_$][A-Za-z0-9_$]*)*\s*\(/;
268
280
  // A file-path shape: a known config/source extension, or a path with a
269
281
  // directory separator. Deliberately requires no spaces — a real path
270
282
  // literal ("`.claude/settings.json`", "`config.yaml`") never has one,
@@ -29,6 +29,37 @@ function editedContentFromEvent(event) {
29
29
  return input.new_source;
30
30
  return null;
31
31
  }
32
+ /**
33
+ * Whether the content contains this literal AS A CALL, not merely as a
34
+ * substring of a longer identifier.
35
+ *
36
+ * Found 2026-09-15 by checking a corpus FAIL rather than assuming it was
37
+ * legitimate: a rule forbidding `fetch()` matched a file containing
38
+ * `_metar_fetch()`. The literal was present verbatim, and entirely the wrong
39
+ * function. The same bare-substring test makes `main()` match `domain()` and
40
+ * `run()` match `rerun()`, and short generic call names are exactly what
41
+ * these rules tend to name.
42
+ *
43
+ * Only the LEADING boundary is checked. The trailing side is already pinned
44
+ * by the pattern itself — every literal reaching this checker ends in an
45
+ * open paren or a call — so requiring a boundary after it would reject the
46
+ * arguments.
47
+ */
48
+ function containsCall(content, pattern) {
49
+ const leadsWithIdentifier = /^[A-Za-z0-9_$]/.test(pattern);
50
+ if (!leadsWithIdentifier)
51
+ return content.includes(pattern);
52
+ let from = 0;
53
+ for (;;) {
54
+ const at = content.indexOf(pattern, from);
55
+ if (at === -1)
56
+ return false;
57
+ const before = at === 0 ? "" : content[at - 1];
58
+ if (!/[A-Za-z0-9_$.]/.test(before))
59
+ return true;
60
+ from = at + 1;
61
+ }
62
+ }
32
63
  export function runCodeContentChecks(classifications, events) {
33
64
  const editedContents = [];
34
65
  for (const event of events) {
@@ -41,7 +72,7 @@ export function runCodeContentChecks(classifications, events) {
41
72
  let foundContent;
42
73
  for (const content of editedContents) {
43
74
  for (const pattern of patterns) {
44
- if (content.includes(pattern)) {
75
+ if (containsCall(content, pattern)) {
45
76
  foundPattern = pattern;
46
77
  foundContent = content;
47
78
  break;
@@ -1,5 +1,32 @@
1
1
  import type { TranscriptEvent, CheckResult } from "../types.js";
2
2
  import type { DeterministicClassification } from "./classify.js";
3
+ /**
4
+ * Appends the canonical long spelling of any short destructive flag the
5
+ * text uses, so a rule banning `git push --force` also catches `git push -f`.
6
+ *
7
+ * Extracted from searchHaystack 2026-09-15 so the pre-execution guard uses
8
+ * exactly the same aliasing as the post-hoc checker. A guard that missed
9
+ * `-f` while the report caught it would be worse than having neither.
10
+ */
11
+ export declare function canonicalise(text: string): string;
12
+ /**
13
+ * Word-boundary-aware match, not a naive substring search — otherwise a
14
+ * pattern like "git push --force" would false-positive on the SAFER
15
+ * "git push --force-with-lease" (caught by an actual failing test before
16
+ * this fix, not assumed).
17
+ *
18
+ * The trailing boundary only applies when the pattern itself ends in a
19
+ * word character — that's the only case where appending more word
20
+ * characters could form a genuinely different, longer token (like
21
+ * "--force" extending into "--force-with-lease"). A pattern that already
22
+ * ends in punctuation (e.g. "http://", ".env") can't be turned into a
23
+ * different token that way, and real occurrences of it (a real URL, a
24
+ * real filename) always have more characters immediately after — a real
25
+ * bug found by testing this against an actual "http://example.com"
26
+ * string: the old unconditional boundary made "http://" unmatchable
27
+ * against any real URL, ever.
28
+ */
29
+ export declare function matchesPattern(haystack: string, pattern: string): boolean;
3
30
  /**
4
31
  * Checks a single deterministic rule against the transcript by scanning
5
32
  * every event for the rule's literal banned pattern(s). No API calls, no
@@ -54,8 +54,19 @@ function searchHaystack(event) {
54
54
  const raw = eventSearchText(event);
55
55
  if (event.kind !== "tool_use")
56
56
  return raw;
57
- const extra = FLAG_ALIASES.filter((a) => a.context.test(raw) && a.short.test(raw)).map((a) => a.canonical);
58
- return extra.length > 0 ? `${raw} ${extra.join(" ")}` : raw;
57
+ return canonicalise(raw);
58
+ }
59
+ /**
60
+ * Appends the canonical long spelling of any short destructive flag the
61
+ * text uses, so a rule banning `git push --force` also catches `git push -f`.
62
+ *
63
+ * Extracted from searchHaystack 2026-09-15 so the pre-execution guard uses
64
+ * exactly the same aliasing as the post-hoc checker. A guard that missed
65
+ * `-f` while the report caught it would be worse than having neither.
66
+ */
67
+ export function canonicalise(text) {
68
+ const extra = FLAG_ALIASES.filter((a) => a.context.test(text) && a.short.test(text)).map((a) => a.canonical);
69
+ return extra.length > 0 ? `${text} ${extra.join(" ")}` : text;
59
70
  }
60
71
  function escapeRegex(literal) {
61
72
  return literal.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
@@ -77,7 +88,7 @@ function escapeRegex(literal) {
77
88
  * string: the old unconditional boundary made "http://" unmatchable
78
89
  * against any real URL, ever.
79
90
  */
80
- function matchesPattern(haystack, pattern) {
91
+ export function matchesPattern(haystack, pattern) {
81
92
  const lastChar = pattern[pattern.length - 1];
82
93
  const needsTrailingBoundary = /[\w-]/.test(lastChar);
83
94
  const suffix = needsTrailingBoundary ? "(?![\\w-])" : "";
@@ -1,5 +1,6 @@
1
1
  import { violation } from "../types.js";
2
2
  import { isProjectPath } from "./projectPaths.js";
3
+ import { withoutHeredocs } from "./shellCommand.js";
3
4
  /**
4
5
  * Third structured-check primitive: only counts real MUTATIONS of a
5
6
  * protected file, never reads of it. Real false-positive this fixes
@@ -43,7 +44,17 @@ function pathPattern(filePath) {
43
44
  * it mutates after that is a scratch file, not the project's.
44
45
  */
45
46
  const CD_INTO_TEMP = /\bcd\s+["']?(?:\/private)?\/(?:tmp|var\/folders)\b|\bcd\s+["']?[^\s"'&|;]*\/(?:scratchpad|node_modules)\b/;
46
- function mutatesPathInBash(command, filePath) {
47
+ /**
48
+ * A path named only inside a heredoc body was not touched by the command
49
+ * that contains it.
50
+ *
51
+ * Real case, 2026-09-15: a command editing landing/index.html through a
52
+ * Python heredoc was reported as modifying `.claude/`, because the HTML it
53
+ * inserts tells readers to put a hook in `.claude/settings.json`. Writing a
54
+ * path into a file is not mutating that path.
55
+ */
56
+ function mutatesPathInBash(rawCommand, filePath) {
57
+ const command = withoutHeredocs(rawCommand);
47
58
  const p = pathPattern(filePath);
48
59
  const mutations = [
49
60
  // rm / rmdir / unlink targeting the path
@@ -0,0 +1,30 @@
1
+ /**
2
+ * Is this literal shaped like a command, rather than a noun?
3
+ *
4
+ * The single most important restriction in the guard, and it exists because
5
+ * the first version was measured before it shipped. Replaying 16,322 real
6
+ * tool calls against the 371 literal prohibitions in the corpus, it refused
7
+ * 62% of them. The literals doing the damage were ordinary words that rules
8
+ * name in passing — `browse`, `mix`, `index.ts`, `scripts:`, `AGENTS.md` —
9
+ * each of which appears in perfectly innocent commands all day.
10
+ *
11
+ * A prohibition worth blocking on names an invocation: `git push --force`,
12
+ * `rm -rf`, `npm publish`, or a bare flag like `--no-verify`. So: it must
13
+ * contain whitespace, or begin with a dash. A single bare word is never
14
+ * enough, which does mean a rule that says only "never use `rm`" is not
15
+ * enforced here. That is the right way round. The report still carries it,
16
+ * and refusing to run someone's command on the strength of one ambiguous
17
+ * word is not a trade worth making.
18
+ */
19
+ export declare function literalIsCommandShaped(literal: string): boolean;
20
+ /**
21
+ * Whether a command that is ABOUT TO RUN performs the forbidden thing.
22
+ *
23
+ * The post-hoc checker cannot answer this and says so: a literal in a
24
+ * transcript may be a violation, a grep, or an explanation, and nothing in
25
+ * the string distinguishes them. Before execution the question is narrower
26
+ * and mostly answerable, because the command is the act. What remains is
27
+ * separating the parts of a compound command that do something from the
28
+ * parts that only look at something, which is what this does.
29
+ */
30
+ export declare function commandRunsLiteral(command: string, literal: string): boolean;
@@ -0,0 +1,127 @@
1
+ import { withoutHeredocs } from "./shellCommand.js";
2
+ import { canonicalise, matchesPattern } from "./deterministicChecks.js";
3
+ /**
4
+ * Commands that read. A banned literal appearing as an argument to one of
5
+ * these is being searched for, printed, or paged — not run.
6
+ *
7
+ * This is the whole reason the post-hoc deterministic checker refuses to
8
+ * report a violation: a text match cannot tell an action from a mention.
9
+ * Before the tool runs, most of that ambiguity is gone — the command IS the
10
+ * action — but not all of it, because a command can still quote a literal
11
+ * while doing something harmless with it. This list is what is left of the
12
+ * problem, and it is deliberately a known list: anything not on it is
13
+ * treated as doing something.
14
+ *
15
+ * `sed` is absent on purpose. `sed -i` edits in place; plain `sed` does not,
16
+ * and the difference is handled below rather than by listing the name.
17
+ */
18
+ const READ_ONLY = new Set([
19
+ "grep", "rg", "ag", "ack", "egrep", "fgrep",
20
+ "echo", "printf", "cat", "bat", "head", "tail", "less", "more",
21
+ "find", "fd", "ls", "wc", "sort", "uniq", "diff", "comm",
22
+ "awk", "jq", "yq", "cut", "tr", "column", "tee",
23
+ "which", "type", "file", "stat", "man", "help",
24
+ ]);
25
+ /** Read-only git subcommands — `git log` cannot delete anything. */
26
+ const READ_ONLY_GIT = new Set(["log", "show", "diff", "status", "blame", "describe", "config", "remote", "branch", "tag", "ls-files", "rev-parse", "shortlog"]);
27
+ /**
28
+ * Splits a shell command into the pieces that run separately.
29
+ *
30
+ * Crude by design: this is not a shell parser and must never pretend to be
31
+ * one. It exists so that `grep "rm -rf" notes.txt && npm run build` is read
32
+ * as two things, one of which searches for a string and one of which does
33
+ * not, rather than as one blob containing a banned literal.
34
+ */
35
+ function segments(command) {
36
+ return withoutHeredocs(command)
37
+ .split(/\n|&&|\|\||[;|]/)
38
+ .map((s) => s.trim())
39
+ .filter((s) => s.length > 0);
40
+ }
41
+ /** The executable a segment invokes, with env assignments and `sudo` skipped. */
42
+ function leadingCommand(segment) {
43
+ const words = segment.split(/\s+/).filter(Boolean);
44
+ let i = 0;
45
+ while (i < words.length && (/^[A-Za-z_][A-Za-z0-9_]*=/.test(words[i]) || words[i] === "sudo" || words[i] === "command" || words[i] === "time"))
46
+ i++;
47
+ const exe = (words[i] ?? "").replace(/^.*\//, "");
48
+ return exe;
49
+ }
50
+ /**
51
+ * A commit message is text the user wrote, not a command being run.
52
+ *
53
+ * Real case in this project's own history: a commit message containing
54
+ * backticked shell examples. A message that quotes a banned command in order
55
+ * to describe it must not be treated as running it.
56
+ */
57
+ function withoutCommitMessage(segment) {
58
+ return segment.replace(/(-m|--message)(=|\s+)(['"])(?:\\.|(?!\3)[\s\S])*\3/g, "$1 <message>");
59
+ }
60
+ /**
61
+ * Does this segment actually DO the forbidden thing?
62
+ *
63
+ * Returns false for a segment that only reads, searches, or prints.
64
+ */
65
+ function segmentRunsLiteral(segment, literal) {
66
+ const exe = leadingCommand(segment);
67
+ if (READ_ONLY.has(exe))
68
+ return false;
69
+ if (exe === "sed" && !/\s-[A-Za-z]*i\b/.test(segment))
70
+ return false;
71
+ if (exe === "git") {
72
+ const sub = segment.split(/\s+/).filter(Boolean)[1] ?? "";
73
+ // A read-only git subcommand cannot be the destructive act — unless the
74
+ // rule names that subcommand itself, which is a real thing people write
75
+ // ("never run `git config user.email`"), so the literal still has to be
76
+ // checked against it. What is skipped is only the case where the literal
77
+ // is some OTHER command quoted inside a read-only one.
78
+ if (READ_ONLY_GIT.has(sub) && !literal.includes(`git ${sub}`))
79
+ return false;
80
+ }
81
+ return matchesPattern(canonicalise(withoutCommitMessage(segment)), literal);
82
+ }
83
+ /**
84
+ * Is this literal shaped like a command, rather than a noun?
85
+ *
86
+ * The single most important restriction in the guard, and it exists because
87
+ * the first version was measured before it shipped. Replaying 16,322 real
88
+ * tool calls against the 371 literal prohibitions in the corpus, it refused
89
+ * 62% of them. The literals doing the damage were ordinary words that rules
90
+ * name in passing — `browse`, `mix`, `index.ts`, `scripts:`, `AGENTS.md` —
91
+ * each of which appears in perfectly innocent commands all day.
92
+ *
93
+ * A prohibition worth blocking on names an invocation: `git push --force`,
94
+ * `rm -rf`, `npm publish`, or a bare flag like `--no-verify`. So: it must
95
+ * contain whitespace, or begin with a dash. A single bare word is never
96
+ * enough, which does mean a rule that says only "never use `rm`" is not
97
+ * enforced here. That is the right way round. The report still carries it,
98
+ * and refusing to run someone's command on the strength of one ambiguous
99
+ * word is not a trade worth making.
100
+ */
101
+ export function literalIsCommandShaped(literal) {
102
+ const t = literal.trim();
103
+ if (t.length === 0)
104
+ return false;
105
+ if (/^-{1,2}[A-Za-z]/.test(t))
106
+ return true;
107
+ // Must BEGIN like a command too, not merely contain a space: `, and` is a
108
+ // real corpus literal and it blocked 118 commands on its own.
109
+ if (!/^[A-Za-z][A-Za-z0-9_.\/-]*(\s|$)/.test(t))
110
+ return false;
111
+ return /\s/.test(t);
112
+ }
113
+ /**
114
+ * Whether a command that is ABOUT TO RUN performs the forbidden thing.
115
+ *
116
+ * The post-hoc checker cannot answer this and says so: a literal in a
117
+ * transcript may be a violation, a grep, or an explanation, and nothing in
118
+ * the string distinguishes them. Before execution the question is narrower
119
+ * and mostly answerable, because the command is the act. What remains is
120
+ * separating the parts of a compound command that do something from the
121
+ * parts that only look at something, which is what this does.
122
+ */
123
+ export function commandRunsLiteral(command, literal) {
124
+ if (!literalIsCommandShaped(literal))
125
+ return false;
126
+ return segments(command).some((s) => segmentRunsLiteral(s, literal));
127
+ }
@@ -0,0 +1,24 @@
1
+ /**
2
+ * Facts about shell command text that more than one checker needs.
3
+ *
4
+ * Created 2026-09-15. The heredoc guard below lived in testCommands.ts, was
5
+ * used only by the test-command matcher, and the file-mutation checker never
6
+ * saw it — so the same class of false accusation was fixed in one reader and
7
+ * left standing in the other. Anything that reasons about what a shell
8
+ * command DID, rather than what it says, belongs here.
9
+ */
10
+ /**
11
+ * Removes heredoc bodies from a shell command.
12
+ *
13
+ * A command that WRITES a test command is not a command that RUNS one.
14
+ * Found 2026-09-14 on a real session: two false failures whose "last test
15
+ * run" was a shell variable assignment. The actual match came from a
16
+ * heredoc further down, writing a demo fixture whose body contains the
17
+ * string `npm test`. The literal was being generated, never executed — and
18
+ * the tool then read its own report output as the failing result.
19
+ *
20
+ * Handles both quoted and bare delimiters, and leaves everything after the
21
+ * closing delimiter intact, because a real test run often follows the
22
+ * heredoc that set the fixture up.
23
+ */
24
+ export declare function withoutHeredocs(command: string): string;
@@ -0,0 +1,43 @@
1
+ /**
2
+ * Facts about shell command text that more than one checker needs.
3
+ *
4
+ * Created 2026-09-15. The heredoc guard below lived in testCommands.ts, was
5
+ * used only by the test-command matcher, and the file-mutation checker never
6
+ * saw it — so the same class of false accusation was fixed in one reader and
7
+ * left standing in the other. Anything that reasons about what a shell
8
+ * command DID, rather than what it says, belongs here.
9
+ */
10
+ /**
11
+ * Removes heredoc bodies from a shell command.
12
+ *
13
+ * A command that WRITES a test command is not a command that RUNS one.
14
+ * Found 2026-09-14 on a real session: two false failures whose "last test
15
+ * run" was a shell variable assignment. The actual match came from a
16
+ * heredoc further down, writing a demo fixture whose body contains the
17
+ * string `npm test`. The literal was being generated, never executed — and
18
+ * the tool then read its own report output as the failing result.
19
+ *
20
+ * Handles both quoted and bare delimiters, and leaves everything after the
21
+ * closing delimiter intact, because a real test run often follows the
22
+ * heredoc that set the fixture up.
23
+ */
24
+ export function withoutHeredocs(command) {
25
+ const lines = command.split("\n");
26
+ const out = [];
27
+ let closing = null;
28
+ for (const line of lines) {
29
+ if (closing !== null) {
30
+ if (line.trim() === closing)
31
+ closing = null;
32
+ continue;
33
+ }
34
+ const open = line.match(/<<-?\s*(?:'([^']+)'|"([^"]+)"|([A-Za-z_][A-Za-z0-9_]*))/);
35
+ if (open) {
36
+ closing = open[1] ?? open[2] ?? open[3];
37
+ out.push(line.slice(0, open.index));
38
+ continue;
39
+ }
40
+ out.push(line);
41
+ }
42
+ return out.join("\n");
43
+ }
@@ -19,21 +19,6 @@ import type { TranscriptEvent } from "../types.js";
19
19
  * wrong.
20
20
  */
21
21
  export declare const TEST_COMMAND: RegExp;
22
- /**
23
- * Removes heredoc bodies from a shell command.
24
- *
25
- * A command that WRITES a test command is not a command that RUNS one.
26
- * Found 2026-09-14 on a real session: two false failures whose "last test
27
- * run" was a shell variable assignment. The actual match came from a
28
- * heredoc further down, writing a demo fixture whose body contains the
29
- * string `npm test`. The literal was being generated, never executed — and
30
- * the tool then read its own report output as the failing result.
31
- *
32
- * Handles both quoted and bare delimiters, and leaves everything after the
33
- * closing delimiter intact, because a real test run often follows the
34
- * heredoc that set the fixture up.
35
- */
36
- export declare function withoutHeredocs(command: string): string;
37
22
  /**
38
23
  * How many times a single shell command invokes a test suite.
39
24
  *
@@ -54,3 +39,4 @@ export declare function withoutHeredocs(command: string): string;
54
39
  export declare function countTestRuns(command: string): number;
55
40
  /** The first test command run in this session, or null if none ran. */
56
41
  export declare function findTestRun(events: TranscriptEvent[]): string | null;
42
+ export { withoutHeredocs } from "./shellCommand.js";
@@ -1,3 +1,4 @@
1
+ import { withoutHeredocs } from "./shellCommand.js";
1
2
  /**
2
3
  * Commands that run a project's test suite.
3
4
  *
@@ -18,40 +19,6 @@
18
19
  * wrong.
19
20
  */
20
21
  export const TEST_COMMAND = /\b(?:npm|pnpm|yarn|bun)\s+(?:run\s+)?(?:test|verify|check|ci)\b|\bnpx\s+(?:vitest|jest|mocha|ava)\b|\b(?:vitest|jest|mocha|pytest|phpunit|rspec|tox)\b|\bcargo\s+test\b|\bgo\s+test\b|\bmvn\s+(?:test|verify)\b|\bgradle\s+test\b|\bdotnet\s+test\b|\bpython\s+-m\s+(?:pytest|unittest)\b/i;
21
- /**
22
- * Removes heredoc bodies from a shell command.
23
- *
24
- * A command that WRITES a test command is not a command that RUNS one.
25
- * Found 2026-09-14 on a real session: two false failures whose "last test
26
- * run" was a shell variable assignment. The actual match came from a
27
- * heredoc further down, writing a demo fixture whose body contains the
28
- * string `npm test`. The literal was being generated, never executed — and
29
- * the tool then read its own report output as the failing result.
30
- *
31
- * Handles both quoted and bare delimiters, and leaves everything after the
32
- * closing delimiter intact, because a real test run often follows the
33
- * heredoc that set the fixture up.
34
- */
35
- export function withoutHeredocs(command) {
36
- const lines = command.split("\n");
37
- const out = [];
38
- let closing = null;
39
- for (const line of lines) {
40
- if (closing !== null) {
41
- if (line.trim() === closing)
42
- closing = null;
43
- continue;
44
- }
45
- const open = line.match(/<<-?\s*(?:'([^']+)'|"([^"]+)"|([A-Za-z_][A-Za-z0-9_]*))/);
46
- if (open) {
47
- closing = open[1] ?? open[2] ?? open[3];
48
- out.push(line.slice(0, open.index));
49
- continue;
50
- }
51
- out.push(line);
52
- }
53
- return out.join("\n");
54
- }
55
22
  /**
56
23
  * How many times a single shell command invokes a test suite.
57
24
  *
@@ -85,3 +52,4 @@ export function findTestRun(events) {
85
52
  }
86
53
  return null;
87
54
  }
55
+ export { withoutHeredocs } from "./shellCommand.js";
package/dist/cli.js CHANGED
@@ -17,6 +17,7 @@ import { runCodeContentChecks } from "./checks/codeContent.js";
17
17
  import { runFileLifecycleChecks } from "./checks/fileLifecycle.js";
18
18
  import { runClaimEvidenceChecks } from "./checks/claimEvidence.js";
19
19
  import { runHook } from "./hook.js";
20
+ import { runGuard } from "./guard.js";
20
21
  import { runJudgmentChecks } from "./checks/judgmentChecks.js";
21
22
  import { generateReport, generateMarkdownReport } from "./report/generateReport.js";
22
23
  import { generateHtmlReport } from "./report/generateHtmlReport.js";
@@ -580,6 +581,12 @@ program
580
581
  .action(async () => {
581
582
  await runHook(needsLlmResult);
582
583
  });
584
+ program
585
+ .command("guard")
586
+ .description("run as a Claude Code PreToolUse hook - refuse a command that breaks a rule, before it runs (payload on stdin)")
587
+ .action(async () => {
588
+ await runGuard();
589
+ });
583
590
  program
584
591
  .command("doctor")
585
592
  .description("List every Claude Code hook and VS Code auto-task on this machine/project, flag anything suspicious")
@@ -0,0 +1,42 @@
1
+ /**
2
+ * A PreToolUse hook that refuses a command before it runs.
3
+ *
4
+ * The Stop hook added on 2026-09-14 catches a finished session. That is too
5
+ * late for the case it most needs to cover: on 2026-04-25 a Cursor agent
6
+ * running Claude Opus 4.6 deleted PocketOS's production database and every
7
+ * volume-level backup in nine seconds, using a Railway token it found that
8
+ * had been created for managing domains. The agent had a rule — "NEVER run
9
+ * destructive/irreversible git commands...unless the user explicitly
10
+ * requests them" — and afterwards quoted it back, observing that what it had
11
+ * done was "far worse than a force push". A report would have described a
12
+ * database that was already gone.
13
+ *
14
+ * What it does NOT do is the thing it was built for. Blocking on a rule's
15
+ * banned command literal was measured before shipping and cut: see the note
16
+ * above `reason`. It refused 62% of 16,336 real commands, and the residue
17
+ * after two rounds of narrowing was still wrong in a way no matcher fixes.
18
+ * PocketOS would not have been stopped by this hook, and saying otherwise
19
+ * would be the exact failure this tool exists to catch.
20
+ *
21
+ * What remains is real and narrower: rules that name a FILE or a BRANCH.
22
+ * "Never modify `.env`", "never touch `migrations/`", "never commit to
23
+ * `main`". The classifier identified those as a path or a ref rather than
24
+ * guessing which backtick was the prohibition, so a refusal can be stated
25
+ * with a reason that holds up.
26
+ *
27
+ * Three properties, in the order they matter:
28
+ *
29
+ * 1. FORBIDDING rules only, answered by a structured checker. Never a
30
+ * judgment rule, never an LLM opinion, never a requirement, and never a
31
+ * bare command literal. Blocking someone's terminal on a guess is not a
32
+ * trade worth making at any hit rate.
33
+ *
34
+ * 2. It fails OPEN. Any error allows the command and writes to stderr. The
35
+ * opposite choice means a bug in this file stops someone from running
36
+ * anything at all, and they would remove the hook within the hour — which
37
+ * leaves them with no guard rather than an imperfect one.
38
+ *
39
+ * 3. It says which rule and why, in the refusal itself, because a block with
40
+ * no reason is indistinguishable from a broken tool.
41
+ */
42
+ export declare function runGuard(): Promise<void>;
package/dist/guard.js ADDED
@@ -0,0 +1,182 @@
1
+ import { loadRules } from "./rules.js";
2
+ import { classifyRules } from "./checks/classify.js";
3
+ import { runCodeContentChecks } from "./checks/codeContent.js";
4
+ import { runFileLifecycleChecks } from "./checks/fileLifecycle.js";
5
+ import { runGitBranchPolicyChecks } from "./checks/gitBranchPolicy.js";
6
+ import { loadOverrides, ruleFingerprint } from "./overrides.js";
7
+ function readStdin() {
8
+ return new Promise((resolve) => {
9
+ let data = "";
10
+ if (process.stdin.isTTY)
11
+ return resolve("");
12
+ process.stdin.setEncoding("utf8");
13
+ process.stdin.on("data", (c) => (data += c));
14
+ process.stdin.on("end", () => resolve(data));
15
+ process.stdin.on("error", () => resolve(""));
16
+ });
17
+ }
18
+ /**
19
+ * The rules this can answer BEFORE the command runs.
20
+ *
21
+ * Deliberately only the forbidding ones. "Always run the tests before
22
+ * committing" cannot be judged from a single proposed call — whether it was
23
+ * already satisfied is a fact about the session, not about this command —
24
+ * and a guard that blocked on it would fire on the first commit of every
25
+ * session. Requirements stay with the report and the Stop hook.
26
+ */
27
+ function forbidRules(cwd) {
28
+ const overrides = loadOverrides(cwd);
29
+ return classifyRules(loadRules(cwd)).filter((c) => {
30
+ if (overrides.get(ruleFingerprint(c.rule))?.decision === "notARule")
31
+ return false;
32
+ return "polarity" in c && c.polarity === "forbid";
33
+ });
34
+ }
35
+ /**
36
+ * Runs the structured checkers against a single PROPOSED action.
37
+ *
38
+ * They already take a list of events and ask what it did, so a one-event
39
+ * list describing what is about to happen is exactly the right input. This
40
+ * is why the guard cannot drift from the report: same checkers, same
41
+ * verdicts, different tense.
42
+ */
43
+ function structuredBlocks(cwd, event) {
44
+ const cls = forbidRules(cwd);
45
+ const of = (k) => cls.filter((c) => c.kind === k);
46
+ const results = [
47
+ ...runCodeContentChecks(of("codeContent"), [event]),
48
+ ...runFileLifecycleChecks(of("fileLifecycle"), [event]),
49
+ ...runGitBranchPolicyChecks(of("gitBranchPolicy"), [event]),
50
+ ];
51
+ return results
52
+ .filter((r) => r.status === "FAIL")
53
+ .map((r) => ({
54
+ rule: { id: r.ruleId, title: r.ruleTitle, text: "", source: r.ruleSource },
55
+ why: r.evidence,
56
+ }));
57
+ }
58
+ /**
59
+ * Literal command bans do NOT block. Measured, then cut.
60
+ *
61
+ * This was the point of the feature and it does not survive its own
62
+ * measurement. Replaying 16,336 real tool calls against every forbidding
63
+ * rule in the corpus, blocking on literals refused 62% of commands. Two
64
+ * restrictions brought that to 1.6% — the literal has to be shaped like an
65
+ * invocation rather than a noun, and the rule has to name exactly one of
66
+ * them — and what remained was still wrong in a way no matcher can fix.
67
+ *
68
+ * A rule titled "Feature Validation" refused `npm run build` 112 times. It
69
+ * forbids running Playwright without asking; it RECOMMENDS `npm run build`.
70
+ * Forbid polarity, one command-shaped literal, and that literal is the
71
+ * approved command. Another rule refused plain `git status`, because its
72
+ * backticks hold both the thing it bans and the thing it suggests instead.
73
+ *
74
+ * Nothing in a rules file marks which backtick is the prohibition. The
75
+ * report can live with that — it says UNCLEAR and a person reads it. A
76
+ * blocker cannot: it would refuse the recommended command with a confident
77
+ * explanation. So command bans stay in the report and the Stop hook, and
78
+ * only the structured checkers, which know what kind of thing they are
79
+ * looking at because the classifier identified a path or a branch, can
80
+ * refuse anything here.
81
+ *
82
+ * The way back to command bans is an explicit opt-in — the user naming which
83
+ * rules may block — not a cleverer guess. That is a design question for
84
+ * whoever asks for it, not a default.
85
+ */
86
+ function reason(blocks) {
87
+ const lines = blocks.map((b) => ` • Rule ${b.rule.id} — ${b.rule.title}\n ${b.why}`);
88
+ const n = blocks.length;
89
+ return (`RuleReceipt blocked this: it breaks ${n === 1 ? "a rule" : `${n} rules`} in CLAUDE.md.\n\n` +
90
+ lines.join("\n\n") +
91
+ `\n\nIf the rule should not apply here, say so to the user and let them decide. Do not work around the rule by rephrasing the command.`);
92
+ }
93
+ /**
94
+ * A PreToolUse hook that refuses a command before it runs.
95
+ *
96
+ * The Stop hook added on 2026-09-14 catches a finished session. That is too
97
+ * late for the case it most needs to cover: on 2026-04-25 a Cursor agent
98
+ * running Claude Opus 4.6 deleted PocketOS's production database and every
99
+ * volume-level backup in nine seconds, using a Railway token it found that
100
+ * had been created for managing domains. The agent had a rule — "NEVER run
101
+ * destructive/irreversible git commands...unless the user explicitly
102
+ * requests them" — and afterwards quoted it back, observing that what it had
103
+ * done was "far worse than a force push". A report would have described a
104
+ * database that was already gone.
105
+ *
106
+ * What it does NOT do is the thing it was built for. Blocking on a rule's
107
+ * banned command literal was measured before shipping and cut: see the note
108
+ * above `reason`. It refused 62% of 16,336 real commands, and the residue
109
+ * after two rounds of narrowing was still wrong in a way no matcher fixes.
110
+ * PocketOS would not have been stopped by this hook, and saying otherwise
111
+ * would be the exact failure this tool exists to catch.
112
+ *
113
+ * What remains is real and narrower: rules that name a FILE or a BRANCH.
114
+ * "Never modify `.env`", "never touch `migrations/`", "never commit to
115
+ * `main`". The classifier identified those as a path or a ref rather than
116
+ * guessing which backtick was the prohibition, so a refusal can be stated
117
+ * with a reason that holds up.
118
+ *
119
+ * Three properties, in the order they matter:
120
+ *
121
+ * 1. FORBIDDING rules only, answered by a structured checker. Never a
122
+ * judgment rule, never an LLM opinion, never a requirement, and never a
123
+ * bare command literal. Blocking someone's terminal on a guess is not a
124
+ * trade worth making at any hit rate.
125
+ *
126
+ * 2. It fails OPEN. Any error allows the command and writes to stderr. The
127
+ * opposite choice means a bug in this file stops someone from running
128
+ * anything at all, and they would remove the hook within the hour — which
129
+ * leaves them with no guard rather than an imperfect one.
130
+ *
131
+ * 3. It says which rule and why, in the refusal itself, because a block with
132
+ * no reason is indistinguishable from a broken tool.
133
+ */
134
+ export async function runGuard() {
135
+ const allow = () => {
136
+ process.stdout.write(JSON.stringify({}));
137
+ };
138
+ try {
139
+ const raw = await readStdin();
140
+ const input = raw ? JSON.parse(raw) : {};
141
+ const cwd = input.cwd || process.cwd();
142
+ const tool = input.tool_name ?? "";
143
+ const toolInput = input.tool_input ?? {};
144
+ if (loadRules(cwd).length === 0)
145
+ return allow();
146
+ let blocks = [];
147
+ if (tool === "Bash" && typeof toolInput.command === "string") {
148
+ const event = {
149
+ role: "assistant", kind: "tool_use", toolName: "Bash",
150
+ input: { command: toolInput.command }, timestamp: "",
151
+ };
152
+ blocks = structuredBlocks(cwd, event);
153
+ }
154
+ else if (tool === "Write" || tool === "Edit" || tool === "NotebookEdit") {
155
+ const event = {
156
+ role: "assistant", kind: "tool_use", toolName: tool,
157
+ input: toolInput, timestamp: "",
158
+ };
159
+ blocks = structuredBlocks(cwd, event);
160
+ }
161
+ else {
162
+ return allow();
163
+ }
164
+ if (blocks.length === 0)
165
+ return allow();
166
+ process.stdout.write(JSON.stringify({
167
+ hookSpecificOutput: {
168
+ hookEventName: "PreToolUse",
169
+ permissionDecision: "deny",
170
+ permissionDecisionReason: reason(blocks),
171
+ },
172
+ }));
173
+ // Exit 2 is what actually blocks the call; the JSON carries the reason.
174
+ process.exitCode = 2;
175
+ }
176
+ catch (err) {
177
+ // Fail open, and never with exit 2 — an exit 2 from a crash would block
178
+ // every command the session tries to run.
179
+ process.stderr.write(`rulereceipt guard: allowing command, check did not complete (${err instanceof Error ? err.message : String(err)})\n`);
180
+ allow();
181
+ }
182
+ }
package/dist/hook.js CHANGED
@@ -22,12 +22,25 @@ function readStdin() {
22
22
  * what to do and repeating it invites the model to argue with the wording
23
23
  * instead of going and running the thing.
24
24
  */
25
- function blockReason(failures) {
26
- const lines = failures.map((f) => ` • Rule ${f.ruleId} — ${f.ruleTitle}\n ${f.evidence}`);
27
- const n = failures.length;
28
- return (`RuleReceipt: ${n} rule${n === 1 ? "" : "s"} in CLAUDE.md ${n === 1 ? "was" : "were"} not followed in this session.\n\n` +
29
- lines.join("\n\n") +
30
- `\n\nDo not report this work as finished until the above is resolved or you have said plainly, to the user, that it is still open and why.`);
25
+ function blockReason(failures, unverified) {
26
+ const parts = [];
27
+ if (failures.length > 0) {
28
+ const n = failures.length;
29
+ parts.push(`RuleReceipt: ${n} rule${n === 1 ? "" : "s"} in CLAUDE.md ${n === 1 ? "was" : "were"} not followed in this session.\n\n` +
30
+ failures.map((f) => ` • Rule ${f.ruleId} — ${f.ruleTitle}\n ${f.evidence}`).join("\n\n"));
31
+ }
32
+ // Worded as a question about evidence rather than as a finding, because
33
+ // that is what it is. Nothing here says the claim is false. It says
34
+ // nothing in this session shows it to be true, and that the difference
35
+ // belongs to the user rather than to the summary.
36
+ if (unverified.length > 0) {
37
+ parts.push(`RuleReceipt: this session claims work is done, and nothing recorded here verifies it.\n\n` +
38
+ unverified.map((u) => ` • Rule ${u.ruleId} — ${u.ruleTitle}\n ${u.evidence}`).join("\n\n") +
39
+ `\n\nThis is not a claim that you are wrong. It may well have been verified somewhere this transcript cannot see.`);
40
+ }
41
+ parts.push(`Do not report this work as finished until the above is resolved, or you have said plainly, to the user, ` +
42
+ `what was actually verified and what was not.`);
43
+ return parts.join("\n\n");
31
44
  }
32
45
  /**
33
46
  * A Stop hook that refuses to let a session end on a broken rule.
@@ -88,9 +101,13 @@ export async function runHook(needsLlmResult) {
88
101
  // it is part of the contract.
89
102
  const { results } = await evaluateSession(cwd, rules, events, false, needsLlmResult);
90
103
  const failures = results.filter((r) => r.status === "FAIL" && r.outcome !== "not_run");
91
- if (failures.length === 0)
104
+ // Deliberately NOT a FAIL. See CheckResult.unverifiedClaim: the report
105
+ // calls this unclear and the gate refuses it, and that is the only place
106
+ // the two are allowed to disagree.
107
+ const unverified = results.filter((r) => r.unverifiedClaim === true);
108
+ if (failures.length === 0 && unverified.length === 0)
92
109
  return void emit({});
93
- return void emit({ decision: "block", reason: blockReason(failures) });
110
+ return void emit({ decision: "block", reason: blockReason(failures, unverified) });
94
111
  }
95
112
  catch (err) {
96
113
  // Property 3: fail open, but never silently.
@@ -117,7 +117,7 @@ export function parseLine(line) {
117
117
  events.push({ role: "assistant", kind: "text", text: b.text, timestamp });
118
118
  }
119
119
  else if (b.type === "tool_use" && typeof b.name === "string") {
120
- events.push({ role: "assistant", kind: "tool_use", toolName: b.name, input: b.input, timestamp });
120
+ events.push({ role: "assistant", kind: "tool_use", toolName: b.name, input: b.input, timestamp, toolUseId: typeof b.id === "string" ? b.id : undefined });
121
121
  }
122
122
  }
123
123
  }
@@ -140,6 +140,7 @@ export function parseLine(line) {
140
140
  content: extractToolResultText(b.content),
141
141
  isError: b.is_error === true,
142
142
  timestamp,
143
+ toolUseId: typeof b.tool_use_id === "string" ? b.tool_use_id : undefined,
143
144
  });
144
145
  }
145
146
  }
package/dist/types.d.ts CHANGED
@@ -16,6 +16,17 @@ export interface TranscriptToolUseEvent {
16
16
  toolName: string;
17
17
  input: unknown;
18
18
  timestamp: string;
19
+ /**
20
+ * The tool_use id, when the transcript carries one.
21
+ *
22
+ * Added 2026-09-16. Without it a result can only be attributed to "the
23
+ * call immediately before", which was documented here as safe on the
24
+ * grounds that every turn contained exactly one tool call. Parallel tool
25
+ * calls are now ordinary — a real session issues three in a row and then
26
+ * one result — and under that shape the positional rule silently drops
27
+ * the test run that a completion claim depended on.
28
+ */
29
+ toolUseId?: string;
19
30
  }
20
31
  export interface TranscriptToolResultEvent {
21
32
  role: "user";
@@ -23,6 +34,8 @@ export interface TranscriptToolResultEvent {
23
34
  content: string;
24
35
  isError: boolean;
25
36
  timestamp: string;
37
+ /** The id of the call this result belongs to, when the transcript has it. */
38
+ toolUseId?: string;
26
39
  }
27
40
  export type TranscriptEvent = TranscriptTextEvent | TranscriptToolUseEvent | TranscriptToolResultEvent;
28
41
  export type CheckStatus = "PASS" | "FAIL" | "UNCLEAR";
@@ -74,6 +87,23 @@ export interface CheckResult {
74
87
  * not blur them.
75
88
  */
76
89
  needsHuman?: boolean;
90
+ /**
91
+ * A completion claim that nothing in this session verified.
92
+ *
93
+ * The one place the report and the gate deliberately disagree. The report
94
+ * renders this UNCLEAR, because absence genuinely proves nothing — the
95
+ * tests may have run in another terminal and a transcript cannot see that.
96
+ * The Stop hook refuses the exit on it anyway, because it is not asserting
97
+ * a violation: it is declining to let "done" end a session when nothing
98
+ * here backs it, and the way out is one sentence saying so out loud.
99
+ *
100
+ * Added 2026-09-16. It is the case anthropics/claude-code#90542 is about
101
+ * from end to end — deploy artifacts produced and the application never
102
+ * opened, a completion whose evidence is a plan, "state what was VERIFIED
103
+ * versus only PLANNED" — and the gate was silent on all of it, firing only
104
+ * when a recorded run CONTRADICTED a claim.
105
+ */
106
+ unverifiedClaim?: boolean;
77
107
  /**
78
108
  * The outcome in the five-value vocabulary. Optional while the checkers
79
109
  * are migrated one at a time; `status` remains the fallback.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "rulereceipt",
3
- "version": "0.1.36",
3
+ "version": "0.1.38",
4
4
  "description": "Checks whether a Claude Code session actually followed your CLAUDE.md / AGENTS.md rules, with evidence.",
5
5
  "repository": {
6
6
  "type": "git",