rulereceipt 0.1.59 → 0.1.60

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -289,6 +289,23 @@ independently confirm it describes the session it claims to.
289
289
  Nothing is uploaded. The file is written to your working directory and
290
290
  goes wherever you choose to send it.
291
291
 
292
+ ## A verdict looks wrong?
293
+
294
+ ```bash
295
+ npx rulereceipt wrong <rule-handle>
296
+ ```
297
+
298
+ Builds a report of that rule, the verdict, how it was decided and the
299
+ session lines around it, with obvious secrets, your home path and email
300
+ addresses masked. It is saved to `.rulereceipt/wrong-<handle>.md` and
301
+ printed so you can read and edit it, along with a link to a pre-filled
302
+ GitHub issue that you open yourself. Nothing is sent. `check` prints the
303
+ command after every report that has a decided verdict, and the HTML report
304
+ has a "Verdict wrong?" link on each one that carries only the version and
305
+ the verdict, never the rule or the evidence.
306
+
307
+ Every accuracy fix in this project has come from a report like this.
308
+
292
309
  ## Exit codes
293
310
 
294
311
  `check` exits **1** when a rule was actually broken, and **0** otherwise,
@@ -379,6 +396,13 @@ npx tsx src/cli.ts demo
379
396
 
380
397
  ## Trust, privacy and licensing
381
398
 
399
+ **What it can't see, it says so.** [KNOWN-GAPS.md](KNOWN-GAPS.md) lists
400
+ exactly where the evidence runs out — commands in another terminal, clicks on
401
+ the permission prompt, `rm`/delete not bound to a rule's subject, edited
402
+ transcripts, IDE sessions without logs. When RuleReceipt hits one of those it
403
+ reports "Can't tell", never a guess. Gaps we know about are safer than gaps we
404
+ don't.
405
+
382
406
  **Nothing leaves your machine unless you ask.** Your code, rules, and
383
407
  session content never leave your computer, ever. Plain `rulereceipt check`
384
408
  makes zero network calls. `--llm`, `--share`, and `--telemetry` are all
package/dist/audit.d.ts CHANGED
@@ -1,3 +1,4 @@
1
+ import { type LoadGraphEntry } from "./rules.js";
1
2
  import type { Rule } from "./types.js";
2
3
  /**
3
4
  * A rules-only health score — how much of a rules file can actually be checked,
@@ -31,3 +32,31 @@ export interface RulesAudit {
31
32
  export declare function auditRules(rules: Rule[]): RulesAudit;
32
33
  /** A short, readable audit. Never says "compliant" — it measures the file, not a session. */
33
34
  export declare function renderAudit(a: RulesAudit, md?: boolean): string;
35
+ /**
36
+ * A file-level problem the audit can name WITHOUT a session: things that stop a
37
+ * rules file from ever reaching the agent as a useful rule. Deliberately about
38
+ * load / shape / checkability only — never "the model ignored you," which this
39
+ * cannot know without a transcript.
40
+ */
41
+ export interface Diagnostic {
42
+ id: "no-rules-file" | "empty-or-pointer" | "zero-rules" | "docs-heavy" | "shadowed-file" | "size-warn" | "template-text" | "broken-import";
43
+ severity: "info" | "warn";
44
+ message: string;
45
+ }
46
+ /** The full doorstep audit for a project: counts + load graph + diagnostics. */
47
+ export interface ProjectAudit extends RulesAudit {
48
+ /** Every candidate rules file the walk saw, loaded or shadowed. */
49
+ loadGraph: LoadGraphEntry[];
50
+ /** File-level problems worth surfacing before any session. */
51
+ diagnostics: Diagnostic[];
52
+ /** How many rules came from Claude Code memory (not a file, so not in loadGraph). */
53
+ memoryRules: number;
54
+ }
55
+ /**
56
+ * The doorstep audit: what loaded, what's checkable, what to fix — no session.
57
+ * `check` proves what a real session did; this answers the day-one questions a
58
+ * cold `npx rulereceipt audit` should, before any transcript exists.
59
+ */
60
+ export declare function auditProject(cwd: string): ProjectAudit;
61
+ /** The doorstep render: load graph → summary → diagnostics → top fixes. */
62
+ export declare function renderProjectAudit(pa: ProjectAudit, md?: boolean): string;
package/dist/audit.js CHANGED
@@ -1,5 +1,8 @@
1
+ import { existsSync, readFileSync } from "node:fs";
2
+ import { dirname, isAbsolute, relative, resolve } from "node:path";
1
3
  import { classifyRules } from "./checks/classify.js";
2
4
  import { adviseRules } from "./checkability.js";
5
+ import { describeRuleSources, loadRules } from "./rules.js";
3
6
  export function auditRules(rules) {
4
7
  let checkable = 0;
5
8
  let judgment = 0;
@@ -57,3 +60,203 @@ export function renderAudit(a, md = false) {
57
60
  out.push("Full advice, rule by rule: rulereceipt rules --advise");
58
61
  return out.join("\n");
59
62
  }
63
+ /**
64
+ * Claude Code reads a bounded prefix of a rules file; past it the rest is
65
+ * silently dropped, so a rule below the cut never loads. The figure is a
66
+ * conservative approximation (character count, not bytes, so non-English files
67
+ * aren't over-counted) and is WARN-only — never a FAIL. TODO: pin to the exact
68
+ * current documented limit before leaning on the number in marketing.
69
+ */
70
+ const SIZE_WARN_CHARS = 40000;
71
+ /** Placeholder text left in from a template — the rule was never actually written. */
72
+ const TEMPLATE_TEXT = /\[your\s+[^\]]+\]|<your\s+[^>]+>|\[project[_ ]name\]|\[TODO\]|^\s*(?:TODO|FIXME)\s*:/im;
73
+ /** Fenced and inline code, masked before scanning for imports/placeholders. */
74
+ function stripCode(text) {
75
+ return text.replace(/```[\s\S]*?```|~~~[\s\S]*?~~~|`[^`\n]*`/g, " ");
76
+ }
77
+ /**
78
+ * `@path` imports the agent follows. Conservative on purpose: only an `@` at a
79
+ * word start (so an email's `@` never counts), and only when the target looks
80
+ * like a file — has a rules-file extension, or starts with `./ ../ /`. That
81
+ * excludes npm scopes like `@types/node` (no extension, no `./`), which are the
82
+ * classic false positive. Resolved relative to the importing file.
83
+ */
84
+ const IMPORT_LINE = /(?:^|[\s(])@((?:\.{0,2}\/)?[\w./-]+\.(?:md|markdown|mdc|txt)|(?:\.{1,2}\/|\/)[\w./-]+)/g;
85
+ function brokenImports(filePath, text) {
86
+ const dir = dirname(filePath);
87
+ const broken = [];
88
+ for (const m of stripCode(text).matchAll(IMPORT_LINE)) {
89
+ const target = m[1];
90
+ const abs = isAbsolute(target) ? target : resolve(dir, target);
91
+ if (!existsSync(abs))
92
+ broken.push(target);
93
+ }
94
+ return broken;
95
+ }
96
+ /** Content that is a pointer to another file, not rules of its own ("see AGENTS.md"). */
97
+ const POINTER = /^\s*(?:#[^\n]*\n)?\s*(?:see|refer\s+to|read|follow|use|check)\b[^\n]{0,80}?\.(?:md|markdown|txt|mdc)\b/i;
98
+ function isPointerFile(path) {
99
+ try {
100
+ const text = readFileSync(path, "utf-8").trim();
101
+ return text.length > 0 && text.length < 220 && POINTER.test(text);
102
+ }
103
+ catch {
104
+ return false;
105
+ }
106
+ }
107
+ function buildDiagnostics(cwd, graph, a) {
108
+ const diags = [];
109
+ const loaded = graph.filter((g) => g.status === "loaded");
110
+ const shadowed = graph.filter((g) => g.status === "shadowed");
111
+ const short = (p) => {
112
+ const r = relative(cwd, p);
113
+ return r && !r.startsWith("..") ? r : p;
114
+ };
115
+ // Nothing to check against at all.
116
+ if (loaded.length === 0 && a.total === 0) {
117
+ diags.push({
118
+ id: "no-rules-file",
119
+ severity: "warn",
120
+ message: "No rules file found. Add a CLAUDE.md or AGENTS.md at the repo root (or .cursor/rules, copilot-instructions.md, .windsurfrules, GEMINI.md) with the rules you want checked.",
121
+ });
122
+ return diags;
123
+ }
124
+ // A shadowed file is present but the agent ignores it — the single most
125
+ // confusing "why isn't my rule firing?" case, so it leads.
126
+ for (const s of shadowed) {
127
+ const ignored = s.ruleCount > 0 ? ` — ${s.ruleCount} rule${s.ruleCount === 1 ? "" : "s"} not applied` : "";
128
+ diags.push({
129
+ id: "shadowed-file",
130
+ severity: "warn",
131
+ message: `${short(s.path)} is present but not loaded (${s.note})${ignored}.`,
132
+ });
133
+ }
134
+ // Loaded files that carry no rules of their own: empty, or just a pointer to
135
+ // another file. A pointer parses to a rule or two ("See AGENTS.md"), so it is
136
+ // caught by content, not only by a zero count.
137
+ for (const l of loaded) {
138
+ if (l.ruleCount === 0) {
139
+ diags.push({ id: "empty-or-pointer", severity: "warn", message: `${short(l.path)} is loaded but has no rules in it yet.` });
140
+ }
141
+ else if (l.ruleCount <= 2 && isPointerFile(l.path)) {
142
+ diags.push({
143
+ id: "empty-or-pointer",
144
+ severity: "warn",
145
+ message: `${short(l.path)} only points to another file — put the actual rules where the agent will read them, or use an @import the agent follows.`,
146
+ });
147
+ }
148
+ }
149
+ // Files exist and load, but nothing parsed as a rule anywhere.
150
+ if (loaded.length > 0 && a.total === 0) {
151
+ diags.push({
152
+ id: "zero-rules",
153
+ severity: "warn",
154
+ message: "Rules files were loaded but none parsed into a rule. Use headings, numbered items, or `-` bullets so each rule stands on its own.",
155
+ });
156
+ }
157
+ // Content-level problems on each loaded file: silent truncation, leftover
158
+ // template text, and imports that resolve to nothing. Read once per file.
159
+ for (const l of loaded) {
160
+ let text = "";
161
+ try {
162
+ text = readFileSync(l.path, "utf-8");
163
+ }
164
+ catch {
165
+ continue;
166
+ }
167
+ if (text.length > SIZE_WARN_CHARS) {
168
+ diags.push({
169
+ id: "size-warn",
170
+ severity: "warn",
171
+ message: `${short(l.path)} is ${text.length.toLocaleString()} characters — large rules files can be silently truncated, so rules near the end may never load. Split it or trim.`,
172
+ });
173
+ }
174
+ if (TEMPLATE_TEXT.test(text)) {
175
+ diags.push({
176
+ id: "template-text",
177
+ severity: "warn",
178
+ message: `${short(l.path)} still has template placeholder text (e.g. "[your project name]", "TODO:") — fill it in or remove it so it reads as a real rule.`,
179
+ });
180
+ }
181
+ const broken = brokenImports(l.path, text);
182
+ if (broken.length > 0) {
183
+ diags.push({
184
+ id: "broken-import",
185
+ severity: "warn",
186
+ message: `${short(l.path)} imports ${broken.slice(0, 3).map((b) => `@${b}`).join(", ")}${broken.length > 3 ? ` (+${broken.length - 3} more)` : ""} — the file doesn't exist, so nothing loads from it.`,
187
+ });
188
+ }
189
+ }
190
+ // A handbook, not a policy: mostly documentation, little to enforce.
191
+ if (a.total >= 8 && a.skipped / a.total > 0.7) {
192
+ diags.push({
193
+ id: "docs-heavy",
194
+ severity: "info",
195
+ message: `${a.skipped} of ${a.total} items are documentation, not rules. That's fine — but for the lines you want enforced, phrase them as Never/Always and put the command, file or branch in \`backticks\`.`,
196
+ });
197
+ }
198
+ return diags;
199
+ }
200
+ /**
201
+ * The doorstep audit: what loaded, what's checkable, what to fix — no session.
202
+ * `check` proves what a real session did; this answers the day-one questions a
203
+ * cold `npx rulereceipt audit` should, before any transcript exists.
204
+ */
205
+ export function auditProject(cwd) {
206
+ const rules = loadRules(cwd);
207
+ const base = auditRules(rules);
208
+ const loadGraph = describeRuleSources(cwd);
209
+ const diagnostics = buildDiagnostics(cwd, loadGraph, base);
210
+ const memoryRules = rules.filter((r) => r.id.startsWith("memory:")).length;
211
+ return { ...base, loadGraph, diagnostics, memoryRules };
212
+ }
213
+ /** The doorstep render: load graph → summary → diagnostics → top fixes. */
214
+ export function renderProjectAudit(pa, md = false) {
215
+ const H = (s) => (md ? `## ${s}` : s);
216
+ const out = [];
217
+ out.push(md ? "# RuleReceipt — rules audit" : "RuleReceipt · rules audit (no session needed)");
218
+ out.push("");
219
+ const loaded = pa.loadGraph.filter((g) => g.status === "loaded");
220
+ const shadowed = pa.loadGraph.filter((g) => g.status === "shadowed");
221
+ out.push(H("Rules files found"));
222
+ if (pa.loadGraph.length === 0) {
223
+ out.push(" (none — no CLAUDE.md / AGENTS.md / Cursor / Copilot / Windsurf / Gemini rules on the path)");
224
+ }
225
+ else {
226
+ for (const g of loaded) {
227
+ out.push(` loaded ${g.format} · ${g.ruleCount} rule${g.ruleCount === 1 ? "" : "s"} (${g.path})`);
228
+ }
229
+ for (const g of shadowed) {
230
+ out.push(` ignored ${g.format} · ${g.note} (${g.path})`);
231
+ }
232
+ if (pa.memoryRules > 0)
233
+ out.push(` loaded Claude memory · ${pa.memoryRules} rule${pa.memoryRules === 1 ? "" : "s"}`);
234
+ }
235
+ out.push("");
236
+ if (pa.checkable + pa.judgment > 0) {
237
+ out.push(H("Can this session be checked against them?"));
238
+ out.push(` ${String(pa.checkable).padStart(4)} checkable — verifiable from a session, no human needed`);
239
+ out.push(` ${String(pa.judgment).padStart(4)} need judgment — a person (or \`--llm\`) decides these`);
240
+ out.push(` ${String(pa.skipped).padStart(4)} documentation — structure/notes, not scored as rules`);
241
+ out.push("");
242
+ out.push(`${pa.percentCheckable}% of your rules can be checked mechanically.`);
243
+ out.push("");
244
+ }
245
+ if (pa.diagnostics.length > 0) {
246
+ out.push(H("What to fix in the setup"));
247
+ for (const d of pa.diagnostics)
248
+ out.push(` ${d.severity === "warn" ? "!" : "·"} ${d.message}`);
249
+ out.push("");
250
+ }
251
+ if (pa.topFixes.length > 0) {
252
+ out.push(H("Top fixes to unlock more checks"));
253
+ for (const f of pa.topFixes) {
254
+ out.push(` • ${f.title.replace(/\s+/g, " ").trim().slice(0, 60)}`);
255
+ out.push(` ${f.suggestion}`);
256
+ }
257
+ out.push("");
258
+ }
259
+ out.push("Full advice, rule by rule: rulereceipt rules --advise");
260
+ out.push("After you run your agent here: rulereceipt check (rule-by-rule proof from the session)");
261
+ return out.join("\n");
262
+ }
@@ -1,3 +1,44 @@
1
1
  import type { ApprovalGateClassification } from "./classify.js";
2
2
  import type { CheckResult, TranscriptEvent } from "../types.js";
3
- export declare function runApprovalGateChecks(classifications: ApprovalGateClassification[], events: TranscriptEvent[]): CheckResult[];
3
+ /**
4
+ * "Never push / commit / open a PR / delete without asking me" — checked per
5
+ * action, not per session.
6
+ *
7
+ * Rewritten 2026-09-28. The first version looked only at the FIRST gated action
8
+ * and cleared the rule if any "shall I…?" appeared anywhere earlier. That
9
+ * passes exactly the case filed most often (anthropics/claude-code #86742,
10
+ * #58079, #67060): one approved push, then a second, unrelated push with no new
11
+ * consent. It also passed "asked, then pushed without waiting".
12
+ *
13
+ * For EACH gated action that actually went through, look at the window since
14
+ * the previous action of the same kind:
15
+ * - the user asked for it, or said yes to an ask -> approved
16
+ * - the tool call was rejected in the prompt -> not an action (skip)
17
+ * - otherwise, could a permission prompt have shown?
18
+ * yes (default / acceptEdits / plan, or mode unknown) -> UNCLEAR: the
19
+ * "Yes" button leaves no trace in the transcript, so a FAIL here
20
+ * could accuse someone who WAS asked and agreed
21
+ * no (bypassPermissions / dontAsk / auto, or the project's allow list
22
+ * covers the command) -> FAIL: nobody was asked
23
+ *
24
+ * Asking is not approval. Claude can say "shall I push?" and push before any
25
+ * reply; the user has to actually say yes. This is the whole reason a bare
26
+ * "no ask found" cannot be a FAIL: the approval may have happened in the
27
+ * permission UI, which the transcript never records.
28
+ */
29
+ type Action = "push" | "commit" | "pr" | "delete";
30
+ /** `Bash(git push:*)`, `Bash(git:*)`, `Bash` style allow entries. */
31
+ export declare function allowListed(command: string, allow: string[]): boolean;
32
+ export interface ApprovalOptions {
33
+ /** `permissions.allow` entries from the project's and user's Claude Code settings. */
34
+ allow?: string[];
35
+ }
36
+ interface Occurrence {
37
+ action: Action;
38
+ command: string;
39
+ verdict: "approved" | "unclear" | "unapproved";
40
+ why: string;
41
+ }
42
+ export declare function approvalOccurrences(events: TranscriptEvent[], actions: Action[], opts?: ApprovalOptions): Occurrence[];
43
+ export declare function runApprovalGateChecks(classifications: ApprovalGateClassification[], events: TranscriptEvent[], opts?: ApprovalOptions): CheckResult[];
44
+ export {};
@@ -1,105 +1,149 @@
1
1
  import { violation } from "../types.js";
2
2
  import { withoutHeredocs } from "./shellCommand.js";
3
- /**
4
- * Did the assistant ASK before doing a thing a rule says needs approval?
5
- *
6
- * From anthropics/claude-code#95494 ("committed without permission", scope
7
- * rewritten and fake results presented) and #92505. A whole cluster of rules
8
- * reads "ask/repeat-back/wait for approval BEFORE you delete | push | commit".
9
- *
10
- * The honest, mechanical half is narrow and stated as such: the rule is
11
- * satisfied by ASKING, so this checks whether the assistant sought approval
12
- * in its recorded text before the action. It does NOT judge whether a reply
13
- * actually granted approval — the "it read my frustration as a yes" case
14
- * (#92505) is a judgment call and stays with the model, not this check.
15
- *
16
- * Bias is deliberately toward NOT accusing: any approval-seeking assistant
17
- * turn before the action clears the rule. A FAIL means the action ran and
18
- * nothing before it asked — the clean case, e.g. a bare `git commit` with no
19
- * "shall I / ok to / before I" anywhere ahead of it.
20
- */
21
- /** Command shapes for each gated action. Kept conservative to avoid accusing. */
22
- // `git\s+…\s+push/commit` allows config flags between the two (`git -c x=y
23
- // push`), which a literal `git push` missed — found by an evasion probe
24
- // 2026-09-22. Up to four intervening tokens; the verb must not be followed by
25
- // a word char so `commit-graph` / `push-option` are not matched.
26
- const ACTION_IN_COMMAND = {
3
+ const IN_COMMAND = {
27
4
  push: /\bgit\s+(?:\S+\s+){0,4}?push(?![\w-])/,
28
5
  commit: /\bgit\s+(?:\S+\s+){0,4}?commit(?![\w-])/,
29
- delete: /\brm\s+-?\w|\bgit\b[^\n]*\s-D\b|\bdrop\s+table\b|\bdelete\s+from\b|\btruncate\b/i,
6
+ pr: /\bgh\s+pr\s+(?:create|merge)\b/,
7
+ delete: /(?:^|[;&|]\s*|\s)rm\s+-?\w|\bgit\b[^\n]*\s-D\b|\bdrop\s+table\b|\bdelete\s+from\b|\btruncate\s+table\b/i,
30
8
  };
31
- /**
32
- * The assistant seeking sign-off. Read generously on purpose: a broad match
33
- * here means fewer false accusations, at the cost of occasionally missing a
34
- * real one — the safe direction for a tool whose whole point is not crying
35
- * wolf.
36
- */
37
- const APPROVAL_SEEK = /\b(?:shall i|should i|may i|can i|do you want|would you like|let me know|before i (?:proceed|do|run|delete|push|commit|go|continue)|your (?:approval|go[- ]?ahead|sign[- ]?off|confirmation)|please confirm|is (?:it|this) ok|ok(?:ay)? to|go ahead\?)\b/i;
38
- function commandOf(event) {
39
- if (event.kind !== "tool_use" || event.toolName !== "Bash")
9
+ const IN_USER_TEXT = {
10
+ push: /\bpush/i,
11
+ commit: /\bcommit/i,
12
+ pr: /\b(?:pr|pull\s+request)\b/i,
13
+ delete: /\b(?:delete|remove|rm|drop|clean\s*up|wipe|get\s+rid)\b/i,
14
+ };
15
+ /** "don't push yet", "no commits", "hold off on the PR": not a yes. */
16
+ function negated(text, verb) {
17
+ const m = text.match(verb);
18
+ if (!m || m.index === undefined)
19
+ return false;
20
+ const before = text.slice(Math.max(0, m.index - 30), m.index);
21
+ return /\b(?:don'?t|do\s+not|no|never|not\s+yet|without|hold\s+off(?:\s+on)?|wait(?:\s+before)?|stop)\b[\s\w]{0,12}$/i.test(before);
22
+ }
23
+ const ASK = /\b(?:shall i|should i|may i|can i|do you want|would you like|want me to|ok(?:ay)? to|is (?:it|this) ok|please confirm|ready to (?:push|commit)|let me know (?:if|when|whether))\b/i;
24
+ const YES = /^\s*(?:y|yes|yep|yeah|yup|sure|ok(?:ay)?|go(?:\s+ahead)?|do\s+it|please\s+do|proceed|lgtm|ship\s+it|sounds\s+good|approved|confirmed|go\s+for\s+it)\b/i;
25
+ const NO_PROMPT_MODES = new Set(["bypassPermissions", "dontAsk", "auto"]);
26
+ function commandOf(e) {
27
+ if (e.kind !== "tool_use" || e.toolName !== "Bash")
40
28
  return "";
41
- const input = event.input;
42
- const raw = input && typeof input.command === "string" ? input.command : "";
43
- // A heredoc that WRITES "git push" into a file is not a push. Strip
44
- // heredoc bodies so only the commands actually invoked are inspected.
45
- return withoutHeredocs(raw);
29
+ const c = e.input?.command;
30
+ // A heredoc that WRITES "git push" into a file is not a push. Strip heredoc
31
+ // bodies so only the commands actually invoked are inspected.
32
+ return typeof c === "string" ? withoutHeredocs(c) : "";
46
33
  }
47
- /**
48
- * The first event index at which one of the gated actions runs, and the
49
- * command that ran it — or null if none did.
50
- */
51
- function firstAction(events, actions) {
52
- const regexes = actions.map((a) => ACTION_IN_COMMAND[a]).filter(Boolean);
53
- for (let i = 0; i < events.length; i++) {
54
- const command = commandOf(events[i]);
55
- if (!command)
34
+ /** The result for the call at `i`: matched by id when present, else the next result. */
35
+ function resultOf(events, i) {
36
+ const call = events[i];
37
+ const id = call.kind === "tool_use" ? call.toolUseId : undefined;
38
+ for (let j = i + 1; j < events.length; j++) {
39
+ const e = events[j];
40
+ if (e.kind !== "tool_result")
56
41
  continue;
57
- if (regexes.some((re) => re.test(command)))
58
- return { index: i, command };
42
+ if (!id || !e.toolUseId || e.toolUseId === id)
43
+ return e;
59
44
  }
60
- return null;
45
+ return undefined;
61
46
  }
62
- /** Did any assistant text before `index` seek approval? */
63
- function askedBefore(events, index) {
64
- for (let i = 0; i < index; i++) {
65
- const e = events[i];
66
- if (e.kind !== "text" || e.role !== "assistant")
67
- continue;
68
- if (APPROVAL_SEEK.test(e.text))
47
+ /** `Bash(git push:*)`, `Bash(git:*)`, `Bash` style allow entries. */
48
+ export function allowListed(command, allow) {
49
+ const c = command.trim();
50
+ return allow.some((entry) => {
51
+ const m = entry.match(/^Bash(?:\((.*)\))?$/);
52
+ if (!m)
53
+ return false;
54
+ if (m[1] === undefined || m[1] === "*" || m[1] === ":*")
69
55
  return true;
56
+ const pat = m[1].replace(/:\*$/, "").replace(/\s*\*$/, "");
57
+ return c === pat || c.startsWith(pat + " ") || (m[1].endsWith("*") && c.startsWith(pat));
58
+ });
59
+ }
60
+ export function approvalOccurrences(events, actions, opts = {}) {
61
+ const out = [];
62
+ const lastIndex = {};
63
+ for (let i = 0; i < events.length; i++) {
64
+ const command = commandOf(events[i]);
65
+ if (!command)
66
+ continue;
67
+ for (const action of actions) {
68
+ if (!IN_COMMAND[action].test(command))
69
+ continue;
70
+ const res = resultOf(events, i);
71
+ if (res && res.kind === "tool_result" && res.isError)
72
+ continue; // rejected in the prompt, or it never went through
73
+ const from = (lastIndex[action] ?? -1) + 1;
74
+ lastIndex[action] = i;
75
+ const window = events.slice(from, i);
76
+ let approved = false;
77
+ let asked = false;
78
+ for (const e of window) {
79
+ if (e.kind !== "text")
80
+ continue;
81
+ if (e.role === "assistant" && ASK.test(e.text))
82
+ asked = true;
83
+ if (e.role !== "user")
84
+ continue;
85
+ if (IN_USER_TEXT[action].test(e.text) && !negated(e.text, IN_USER_TEXT[action]))
86
+ approved = true;
87
+ else if (asked && YES.test(e.text))
88
+ approved = true;
89
+ }
90
+ const short = command.replace(/\s+/g, " ").trim().slice(0, 80);
91
+ if (approved) {
92
+ out.push({ action, command: short, verdict: "approved", why: "the user asked for it or said yes since the last one" });
93
+ continue;
94
+ }
95
+ const call = events[i];
96
+ const mode = call.kind === "tool_use" ? call.permissionMode : undefined;
97
+ const listed = allowListed(command, opts.allow ?? []);
98
+ const noPrompt = (mode && NO_PROMPT_MODES.has(mode)) || listed;
99
+ if (action === "delete") {
100
+ // `rm` / `drop` / `delete` are far too common and generic to bind
101
+ // confidently to a specific rule's subject: a test-cleanup `rm -rf
102
+ // /tmp/x` is not "wiping the production database", but the command
103
+ // matcher can't tell them apart. So a delete gate NEVER produces a
104
+ // FAIL — it reports UNCLEAR, and the guard still asks before the call.
105
+ // Found 2026-09-28: a "Never wipe data storage databases" rule FAILed
106
+ // on an unrelated temp-dir `rm -rf` once the one-engine fix made
107
+ // approval-gate rules visible in `check`. push/commit/pr are specific
108
+ // git/gh operations and keep the FAIL path.
109
+ out.push({ action, command: short, verdict: "unclear", why: "a delete/rm command ran; a transcript can't tell whether it hit the data this rule protects, or whether it was approved" });
110
+ }
111
+ else if (noPrompt) {
112
+ out.push({ action, command: short, verdict: "unapproved", why: listed ? "the command is on the allow list, so no prompt was shown" : `permission mode was ${mode}, so no prompt was shown` });
113
+ }
114
+ else {
115
+ out.push({ action, command: short, verdict: "unclear", why: asked ? "Claude asked but ran it before any reply in the chat" : "nothing in the chat approved it" });
116
+ }
117
+ }
70
118
  }
71
- return false;
119
+ return out;
72
120
  }
73
- export function runApprovalGateChecks(classifications, events) {
121
+ export function runApprovalGateChecks(classifications, events, opts = {}) {
74
122
  return classifications.map(({ rule, actions, polarity }) => {
75
- const action = firstAction(events, actions);
76
- if (!action) {
77
- return {
78
- ruleId: rule.id,
79
- ruleTitle: rule.title,
80
- ruleSource: rule.source,
81
- status: "UNCLEAR",
82
- outcome: "not_applicable",
123
+ const occ = approvalOccurrences(events, actions, opts);
124
+ const base = { ruleId: rule.id, ruleTitle: rule.title, ruleSource: rule.source, method: "approval_gate" };
125
+ if (occ.length === 0) {
126
+ return { ...base, status: "UNCLEAR", outcome: "not_applicable", evidence: `no ${actions.join("/")} went through this session, so the rule never applied` };
127
+ }
128
+ const bad = occ.find((o) => o.verdict === "unapproved");
129
+ if (bad) {
130
+ const n = occ.filter((o) => o.verdict === "unapproved").length;
131
+ return violation(rule, polarity, `ran "${bad.command}" with no approval: ${bad.why}${n > 1 ? ` (${n} times this session)` : ""}`, {
83
132
  method: "approval_gate",
84
- evidence: `no ${actions.join("/")} action ran this session, so the gate never applied`,
85
- };
133
+ ceiling: "per action: approval is a user message asking for it or a yes after Claude asked, since the previous action of the same kind",
134
+ });
86
135
  }
87
- if (askedBefore(events, action.index)) {
136
+ const unsure = occ.find((o) => o.verdict === "unclear");
137
+ if (unsure) {
88
138
  return {
89
- ruleId: rule.id,
90
- ruleTitle: rule.title,
91
- ruleSource: rule.source,
92
- status: "PASS",
93
- outcome: "pass",
94
- method: "approval_gate",
95
- evidence: "the assistant sought approval before the action",
96
- ceiling: "confirms the assistant ASKED; it does not judge whether the reply granted approval, and cannot see an approval given through the permission UI",
139
+ ...base,
140
+ status: "UNCLEAR",
141
+ outcome: "inconclusive",
142
+ reason: "approval_not_visible",
143
+ ceiling: "a click on the permission prompt leaves no trace in the transcript",
144
+ evidence: `ran "${unsure.command}": ${unsure.why}. Claude Code may have shown a permission prompt you approved; the transcript can't show that. Turn on the rulereceipt guard to force a prompt next time.`,
97
145
  };
98
146
  }
99
- const excerpt = action.command.replace(/\s+/g, " ").trim().slice(0, 70);
100
- return violation(rule, polarity, `the assistant ran "${excerpt}" with no approval sought beforehand`, {
101
- method: "approval_gate",
102
- ceiling: "flags an action taken with no approval-seeking text before it; it cannot see an approval given through the permission UI, so this is the clean no-ask case only",
103
- });
147
+ return { ...base, status: "PASS", outcome: "pass", evidence: `${occ.length} ${actions.join("/")} action(s), each asked for or approved in the chat first` };
104
148
  });
105
149
  }
@@ -154,6 +154,19 @@ export declare function isEventRecord(rule: Rule): boolean;
154
154
  * is the forbidden command ("Never run: `rm -rf /`") is still a rule.
155
155
  */
156
156
  export declare function isCommandDocumentation(rule: Rule): boolean;
157
+ /**
158
+ * Which gated actions a rule makes conditional on the user's say-so.
159
+ *
160
+ * Rewritten 2026-09-28, sentence by sentence. Three shapes:
161
+ * "never|don't <action> … without permission | unless I ask | until the user confirms"
162
+ * "ask|confirm|wait for approval … before <action>"
163
+ * "only <action> when|if|after … asked|told|approved"
164
+ * The action and the consent phrase must sit in the SAME sentence. The old
165
+ * version needed only a signal word somewhere in the rule and an action word
166
+ * somewhere in the rule, so a section that said "ask before preserving compat"
167
+ * and elsewhere "delete the old one" became a delete gate — 16 wrong FAILs
168
+ * removed by requiring them together.
169
+ */
157
170
  export declare function approvalGateActions(rule: Rule): string[];
158
171
  export declare function classifyRule(rule: Rule): Classification;
159
172
  export declare function classifyRules(rules: Rule[]): Classification[];