rulereceipt 0.1.48 → 0.1.50

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -393,6 +393,34 @@ report to the path you name, and `rules --include/--exclude` records a
393
393
  correction in `.rulereceipt/overrides.json`. Plain `rulereceipt check` writes
394
394
  nothing and makes no network calls.
395
395
 
396
+ **Severity, per rule.** A committed, team-shared `.rulereceipt/config.json`
397
+ sets how hard each rule bites in CI, by its stable handle (from `rulereceipt
398
+ rules --list`):
399
+
400
+ ```json
401
+ {
402
+ "rules": {
403
+ "a1b2c3": "off", // hidden from the report, never gates
404
+ "d4e5f6": "warn", // shown, but does not fail the build
405
+ "97h8i9": "error" // shown, FAILS the build — the default for a checkable rule
406
+ },
407
+ "checks": {
408
+ "emoji": "off", // silence a whole check type by name
409
+ "git": "warn" // emoji, attribution, approval, git, files, code, claim, tests, judgment
410
+ }
411
+ }
412
+ ```
413
+
414
+ A per-rule `rules` entry wins over a per-check `checks` entry, which wins over
415
+ the default. No config means today's behaviour: every checkable FAIL is an
416
+ `error`. This is
417
+ the one place severity lives — a team marks the must-not-break rules `error`
418
+ and the nice-to-haves `warn`, so CI gates on what matters instead of going red
419
+ on day one. (Refusing a command *before* it runs is separate, and stays with
420
+ the guard's `rules --forbid` clause-mark — a config that could block on any
421
+ rule would refuse far too much.) The older `{"warn": ["a1b2c3"]}` list still
422
+ works and means the same as `"warn"` above.
423
+
396
424
  **You can verify the package came from this source.** Every release from
397
425
  0.1.19 on is built and published by GitHub Actions and signed with
398
426
  [npm provenance](https://docs.npmjs.com/generating-provenance-statements),
package/dist/cli.js CHANGED
@@ -6,7 +6,7 @@ import { join, dirname, resolve, isAbsolute } from "node:path";
6
6
  import { existsSync, readFileSync, writeFileSync } from "node:fs";
7
7
  import { fileURLToPath } from "node:url";
8
8
  import { parseClaudeMd } from "./parsers/readClaudeMd.js";
9
- import { readLatestTranscript, readTranscriptFromFile, findLatestSessionFile } from "./parsers/transcriptParser.js";
9
+ import { readLatestTranscript, readTranscriptFromFile, findLatestSessionFile, subagentNote } from "./parsers/transcriptParser.js";
10
10
  import { loadRules } from "./rules.js";
11
11
  import { adviseRules } from "./checkability.js";
12
12
  import { shadowedAgentsMd } from "./shadowedAgents.js";
@@ -33,7 +33,7 @@ import { maybeShowWhatsNew } from "./whatsNew.js";
33
33
  import { verifyReceipt, parseReceipt } from "./receipt.js";
34
34
  import { buildBadge } from "./badge.js";
35
35
  import { buildInitGuidance } from "./init.js";
36
- import { loadProjectConfig, handleMap, blockingFailures, warningFailures, PROJECT_CONFIG_PATH } from "./projectConfig.js";
36
+ import { loadProjectConfig, handleMap, blockingFailures, warningFailures, visibleResults, PROJECT_CONFIG_PATH } from "./projectConfig.js";
37
37
  import { maybeCheckUpdates, isUpdateCheckEnabled } from "./updateCheck.js";
38
38
  import { generateDigest } from "./digest.js";
39
39
  import { enableSchedule, disableSchedule, scheduleStatus } from "./schedule.js";
@@ -264,12 +264,14 @@ async function runCheck(opts) {
264
264
  // sends only a random install ID, never rule text or transcript content,
265
265
  // regardless of --llm.
266
266
  const judgmentResults = llm ? await runJudgmentChecks(judgment, events) : judgment.map(({ rule }) => needsLlmResult(rule));
267
- const results = [...deterministicResults, ...judgmentResults];
268
- // Severity: rules a team marked as warnings in .rulereceipt/config.json are
269
- // still reported but do not fail the build. handleFor maps a result back to
270
- // its stable handle so the mark survives edits that renumber rule ids.
267
+ const rawResults = [...deterministicResults, ...judgmentResults];
268
+ // Severity ladder from .rulereceipt/config.json (per rule handle): `off`
269
+ // rules are hidden entirely, `warn` rules are shown but do not fail the
270
+ // build, everything else is `error` (the default). handleFor maps a result
271
+ // back to its stable handle so the mark survives edits that renumber ids.
271
272
  const projectConfig = loadProjectConfig(cwd);
272
273
  const handleFor = handleMap(rules);
274
+ const results = visibleResults(rawResults, projectConfig, handleFor);
273
275
  const blockingFails = blockingFailures(results, projectConfig, handleFor);
274
276
  const warnedFails = warningFailures(results, projectConfig, handleFor);
275
277
  const meta = { sessionFilePath, ruleCount: results.length };
@@ -281,6 +283,9 @@ async function runCheck(opts) {
281
283
  }
282
284
  else {
283
285
  console.log(reportText);
286
+ const subNote = subagentNote(sessionFilePath);
287
+ if (subNote)
288
+ console.log(`\n${subNote}`);
284
289
  }
285
290
  // Shown only to someone who has just read their own broken rules, and only
286
291
  // if they have not already wired it up. See report/gateOffer.ts.
package/dist/guard.d.ts CHANGED
@@ -1,3 +1,8 @@
1
+ import type { Rule } from "./types.js";
2
+ interface Block {
3
+ rule: Rule;
4
+ why: string;
5
+ }
1
6
  /**
2
7
  * A PreToolUse hook that refuses a command before it runs.
3
8
  *
@@ -39,4 +44,27 @@
39
44
  * 3. It says which rule and why, in the refusal itself, because a block with
40
45
  * no reason is indistinguishable from a broken tool.
41
46
  */
47
+ export interface GuardDecision {
48
+ /** True when the proposed call breaks a rule and should be refused. */
49
+ deny: boolean;
50
+ /** The message for the model, or "" when allowed. */
51
+ reason: string;
52
+ /** The rules that would refuse it, empty when allowed. */
53
+ blocks: Block[];
54
+ }
55
+ /**
56
+ * The allow/deny decision for one proposed tool call, with no I/O.
57
+ *
58
+ * Extracted from runGuard 2026-09-25 so the decision is a pure function the
59
+ * shell hook calls today and any future host — an in-process function hook,
60
+ * a different runtime — calls tomorrow without re-deriving the logic. The
61
+ * function-hook interface is still an unshipped Anthropic proposal (#91870),
62
+ * so nothing here targets it; this is only the seam that keeps the adapter
63
+ * thin when it lands. Same reasoning as evaluateSession: one body of code so
64
+ * two callers can never disagree about whether a rule was broken.
65
+ */
66
+ export declare function guardDecision(cwd: string, toolName: string, toolInput: {
67
+ command?: unknown;
68
+ } & Record<string, unknown>): GuardDecision;
42
69
  export declare function runGuard(): Promise<void>;
70
+ export {};
package/dist/guard.js CHANGED
@@ -143,46 +143,42 @@ function reason(blocks) {
143
143
  `\n\nIf the rule should not apply here, say so to the user and let them decide. Do not work around the rule by rephrasing the command.`);
144
144
  }
145
145
  /**
146
- * A PreToolUse hook that refuses a command before it runs.
147
- *
148
- * The Stop hook added on 2026-09-14 catches a finished session. That is too
149
- * late for the case it most needs to cover: on 2026-04-25 a Cursor agent
150
- * running Claude Opus 4.6 deleted PocketOS's production database and every
151
- * volume-level backup in nine seconds, using a Railway token it found that
152
- * had been created for managing domains. The agent had a rule — "NEVER run
153
- * destructive/irreversible git commands...unless the user explicitly
154
- * requests them" — and afterwards quoted it back, observing that what it had
155
- * done was "far worse than a force push". A report would have described a
156
- * database that was already gone.
157
- *
158
- * What it does NOT do is the thing it was built for. Blocking on a rule's
159
- * banned command literal was measured before shipping and cut: see the note
160
- * above `reason`. It refused 62% of 16,336 real commands, and the residue
161
- * after two rounds of narrowing was still wrong in a way no matcher fixes.
162
- * PocketOS would not have been stopped by this hook, and saying otherwise
163
- * would be the exact failure this tool exists to catch.
164
- *
165
- * What remains is real and narrower: rules that name a FILE or a BRANCH.
166
- * "Never modify `.env`", "never touch `migrations/`", "never commit to
167
- * `main`". The classifier identified those as a path or a ref rather than
168
- * guessing which backtick was the prohibition, so a refusal can be stated
169
- * with a reason that holds up.
170
- *
171
- * Three properties, in the order they matter:
172
- *
173
- * 1. FORBIDDING rules only, answered by a structured checker. Never a
174
- * judgment rule, never an LLM opinion, never a requirement, and never a
175
- * bare command literal. Blocking someone's terminal on a guess is not a
176
- * trade worth making at any hit rate.
177
- *
178
- * 2. It fails OPEN. Any error allows the command and writes to stderr. The
179
- * opposite choice means a bug in this file stops someone from running
180
- * anything at all, and they would remove the hook within the hour — which
181
- * leaves them with no guard rather than an imperfect one.
182
- *
183
- * 3. It says which rule and why, in the refusal itself, because a block with
184
- * no reason is indistinguishable from a broken tool.
146
+ * The allow/deny decision for one proposed tool call, with no I/O.
147
+ *
148
+ * Extracted from runGuard 2026-09-25 so the decision is a pure function the
149
+ * shell hook calls today and any future host — an in-process function hook,
150
+ * a different runtime — calls tomorrow without re-deriving the logic. The
151
+ * function-hook interface is still an unshipped Anthropic proposal (#91870),
152
+ * so nothing here targets it; this is only the seam that keeps the adapter
153
+ * thin when it lands. Same reasoning as evaluateSession: one body of code so
154
+ * two callers can never disagree about whether a rule was broken.
185
155
  */
156
+ export function guardDecision(cwd, toolName, toolInput) {
157
+ const allow = { deny: false, reason: "", blocks: [] };
158
+ if (loadRules(cwd).length === 0)
159
+ return allow;
160
+ let blocks = [];
161
+ if (toolName === "Bash" && typeof toolInput.command === "string") {
162
+ const event = {
163
+ role: "assistant", kind: "tool_use", toolName: "Bash",
164
+ input: { command: toolInput.command }, timestamp: "",
165
+ };
166
+ blocks = [...structuredBlocks(cwd, event), ...ratifiedLiteralBlocks(cwd, toolInput.command)];
167
+ }
168
+ else if (toolName === "Write" || toolName === "Edit" || toolName === "NotebookEdit") {
169
+ const event = {
170
+ role: "assistant", kind: "tool_use", toolName,
171
+ input: toolInput, timestamp: "",
172
+ };
173
+ blocks = structuredBlocks(cwd, event);
174
+ }
175
+ else {
176
+ return allow;
177
+ }
178
+ if (blocks.length === 0)
179
+ return allow;
180
+ return { deny: true, reason: reason(blocks), blocks };
181
+ }
186
182
  export async function runGuard() {
187
183
  const allow = () => {
188
184
  process.stdout.write(JSON.stringify({}));
@@ -193,29 +189,10 @@ export async function runGuard() {
193
189
  const cwd = input.cwd || process.cwd();
194
190
  const tool = input.tool_name ?? "";
195
191
  const toolInput = input.tool_input ?? {};
196
- if (loadRules(cwd).length === 0)
197
- return allow();
198
- let blocks = [];
199
- if (tool === "Bash" && typeof toolInput.command === "string") {
200
- const event = {
201
- role: "assistant", kind: "tool_use", toolName: "Bash",
202
- input: { command: toolInput.command }, timestamp: "",
203
- };
204
- blocks = [...structuredBlocks(cwd, event), ...ratifiedLiteralBlocks(cwd, toolInput.command)];
205
- }
206
- else if (tool === "Write" || tool === "Edit" || tool === "NotebookEdit") {
207
- const event = {
208
- role: "assistant", kind: "tool_use", toolName: tool,
209
- input: toolInput, timestamp: "",
210
- };
211
- blocks = structuredBlocks(cwd, event);
212
- }
213
- else {
214
- return allow();
215
- }
216
- if (blocks.length === 0)
192
+ const decision = guardDecision(cwd, tool, toolInput);
193
+ if (!decision.deny)
217
194
  return allow();
218
- const why = reason(blocks);
195
+ const why = decision.reason;
219
196
  process.stdout.write(JSON.stringify({
220
197
  hookSpecificOutput: {
221
198
  hookEventName: "PreToolUse",
@@ -36,4 +36,10 @@ export declare function readTranscriptFromFile(filePath: string): TranscriptEven
36
36
  * false negative — the safe direction for a tool that must not over-accuse.
37
37
  */
38
38
  export declare function findSubagentFiles(sessionFile: string): string[];
39
+ /**
40
+ * A one-line note for the report so a user knows their subagents were included
41
+ * in the check — otherwise the coverage is invisible and they might think a
42
+ * violation a subagent committed went unseen. Null when there were none.
43
+ */
44
+ export declare function subagentNote(sessionFile: string | null): string | null;
39
45
  export declare function readLatestTranscript(cwd: string): TranscriptEvent[];
@@ -197,6 +197,19 @@ export function findSubagentFiles(sessionFile) {
197
197
  const subagentDir = join(dirname(sessionFile), sessionId, "subagents");
198
198
  return listSessionFiles(subagentDir);
199
199
  }
200
+ /**
201
+ * A one-line note for the report so a user knows their subagents were included
202
+ * in the check — otherwise the coverage is invisible and they might think a
203
+ * violation a subagent committed went unseen. Null when there were none.
204
+ */
205
+ export function subagentNote(sessionFile) {
206
+ if (!sessionFile)
207
+ return null;
208
+ const n = findSubagentFiles(sessionFile).length;
209
+ if (n === 0)
210
+ return null;
211
+ return `Checked ${n} subagent session${n === 1 ? "" : "s"} alongside the main one — their actions are held to the same rules.`;
212
+ }
200
213
  export function readLatestTranscript(cwd) {
201
214
  const filePath = findLatestSessionFile(cwd);
202
215
  if (!filePath)
@@ -1,25 +1,59 @@
1
- import type { CheckResult, Rule } from "./types.js";
1
+ import type { CheckResult, CheckMethod, Rule } from "./types.js";
2
2
  /**
3
3
  * A committed, team-shared config at .rulereceipt/config.json.
4
4
  *
5
- * Today it holds one thing: rule handles to treat as WARNINGS — a broken
6
- * "warning" rule is still reported, but it does not fail the build. This is
7
- * the honest version of "severity": a team marks the must-not-break rules as
8
- * errors (the default) and the nice-to-have ones as warnings, so CI gates on
9
- * what actually matters instead of going red on day one.
5
+ * ONE place for "how hard should this rule bite", a ladder per rule handle:
6
+ *
7
+ * off hidden from the report, never gates CI
8
+ * warn shown, does not fail the build
9
+ * error shown, FAILS the build (the default for a checkable rule)
10
+ *
11
+ * This is the honest version of "severity": a team marks must-not-break rules
12
+ * as errors (the default) and nice-to-haves as warnings, and silences the
13
+ * irrelevant ones, so CI gates on what actually matters instead of going red
14
+ * on day one. It replaces three scattered mechanisms — the old `warn` list,
15
+ * `rules --exclude` (now `off`), and the plain default — with one field.
16
+ *
17
+ * Pre-run BLOCKING is deliberately NOT a mode here: refusing a command before
18
+ * it runs still goes through the guard's clause-mark (`rules --forbid`),
19
+ * because a config that could block on any rule would refuse the 62.8% of
20
+ * commands the measured guard already showed it must not. This file governs
21
+ * the report and the CI gate; the guard governs refusal.
10
22
  *
11
23
  * Handles, not rule ids: an id is positional and renumbers when the file is
12
24
  * edited above it; a handle is a content hash, so it survives edits. Get one
13
25
  * from `rulereceipt rules --list`.
26
+ *
27
+ * Backward compatible: the old top-level `warn: [handle, ...]` list still
28
+ * works and means the same as `rules: { <handle>: "warn" }`.
14
29
  */
30
+ export type RuleMode = "off" | "warn" | "error";
15
31
  export interface ProjectConfig {
16
32
  warn: string[];
33
+ rules: Record<string, RuleMode>;
34
+ /** A whole check type, by friendly name (see CHECK_ALIASES) or raw method. */
35
+ checks: Record<string, RuleMode>;
17
36
  }
37
+ /**
38
+ * Friendly names for a check type, so `checks: { "emoji": "off" }` silences
39
+ * every emoji verdict without listing each rule. A result carries a `method`
40
+ * (e.g. "emoji_output"); these are the human-facing aliases for those.
41
+ */
42
+ export declare const CHECK_ALIASES: Record<string, CheckMethod>;
18
43
  export declare const PROJECT_CONFIG_PATH: string;
19
44
  export declare function loadProjectConfig(cwd: string): ProjectConfig;
45
+ /**
46
+ * The mode for one result. Precedence: a per-rule `rules` entry wins over a
47
+ * per-check `checks` entry, which wins over the legacy `warn` list; anything
48
+ * unlisted is `error`, so the default is unchanged and no config means today's
49
+ * behaviour exactly.
50
+ */
51
+ export declare function modeForResult(result: CheckResult, config: ProjectConfig, handleFor: (r: CheckResult) => string): RuleMode;
52
+ /** Results the report should show — everything except rules set to `off`. */
53
+ export declare function visibleResults(results: CheckResult[], config: ProjectConfig, handleFor: (r: CheckResult) => string): CheckResult[];
20
54
  /** A lookup from a result back to its stable rule handle, built from the loaded rules. */
21
55
  export declare function handleMap(rules: Rule[]): (r: CheckResult) => string;
22
- /** FAILs that are NOT configured as warnings — these fail the build. */
56
+ /** FAILs at `error` mode — these fail the build. (`off` never reaches here.) */
23
57
  export declare function blockingFailures(results: CheckResult[], config: ProjectConfig, handleFor: (r: CheckResult) => string): CheckResult[];
24
- /** FAILs that ARE configured as warnings — shown, but they do not fail the build. */
58
+ /** FAILs at `warn` mode — shown, but they do not fail the build. */
25
59
  export declare function warningFailures(results: CheckResult[], config: ProjectConfig, handleFor: (r: CheckResult) => string): CheckResult[];
@@ -1,19 +1,81 @@
1
1
  import { readFileSync } from "node:fs";
2
2
  import { join } from "node:path";
3
3
  import { ruleFingerprint } from "./overrides.js";
4
+ const VALID_MODES = ["off", "warn", "error"];
5
+ /**
6
+ * Friendly names for a check type, so `checks: { "emoji": "off" }` silences
7
+ * every emoji verdict without listing each rule. A result carries a `method`
8
+ * (e.g. "emoji_output"); these are the human-facing aliases for those.
9
+ */
10
+ export const CHECK_ALIASES = {
11
+ emoji: "emoji_output",
12
+ attribution: "attribution_scan",
13
+ approval: "approval_gate",
14
+ git: "git_events",
15
+ files: "file_events",
16
+ code: "code_content",
17
+ claim: "claim_vs_evidence",
18
+ tests: "edit_test_pairing",
19
+ text: "text_scan",
20
+ judgment: "model_judgment",
21
+ };
4
22
  export const PROJECT_CONFIG_PATH = join(".rulereceipt", "config.json");
5
23
  export function loadProjectConfig(cwd) {
6
24
  try {
7
25
  const parsed = JSON.parse(readFileSync(join(cwd, PROJECT_CONFIG_PATH), "utf-8"));
8
- const warn = parsed?.warn;
9
- return { warn: Array.isArray(warn) ? warn.filter((x) => typeof x === "string") : [] };
26
+ const warn = Array.isArray(parsed?.warn) ? parsed.warn.filter((x) => typeof x === "string") : [];
27
+ const readModes = (raw) => {
28
+ const out = {};
29
+ if (raw && typeof raw === "object" && !Array.isArray(raw)) {
30
+ for (const [key, mode] of Object.entries(raw)) {
31
+ if (typeof mode === "string" && VALID_MODES.includes(mode))
32
+ out[key] = mode;
33
+ }
34
+ }
35
+ return out;
36
+ };
37
+ return { warn, rules: readModes(parsed?.rules), checks: readModes(parsed?.checks) };
10
38
  }
11
39
  catch {
12
40
  // Missing or malformed config means no severities configured, never an
13
41
  // error — same fail-open discipline as the rest of the tool.
14
- return { warn: [] };
42
+ return { warn: [], rules: {}, checks: {} };
15
43
  }
16
44
  }
45
+ /** The mode a `checks` entry sets for a result's method, if any. */
46
+ function checkMode(method, config) {
47
+ if (!method || !config.checks || Object.keys(config.checks).length === 0)
48
+ return undefined;
49
+ // A raw method key ("emoji_output") wins over its friendly alias ("emoji").
50
+ if (config.checks[method])
51
+ return config.checks[method];
52
+ for (const [alias, m] of Object.entries(CHECK_ALIASES)) {
53
+ if (m === method && config.checks[alias])
54
+ return config.checks[alias];
55
+ }
56
+ return undefined;
57
+ }
58
+ /**
59
+ * The mode for one result. Precedence: a per-rule `rules` entry wins over a
60
+ * per-check `checks` entry, which wins over the legacy `warn` list; anything
61
+ * unlisted is `error`, so the default is unchanged and no config means today's
62
+ * behaviour exactly.
63
+ */
64
+ export function modeForResult(result, config, handleFor) {
65
+ const handle = handleFor(result);
66
+ if (config.rules[handle])
67
+ return config.rules[handle];
68
+ const byCheck = checkMode(result.method, config);
69
+ if (byCheck)
70
+ return byCheck;
71
+ if (config.warn.includes(handle))
72
+ return "warn";
73
+ return "error";
74
+ }
75
+ /** Results the report should show — everything except rules set to `off`. */
76
+ export function visibleResults(results, config, handleFor) {
77
+ return results.filter((r) => modeForResult(r, config, handleFor) !== "off");
78
+ }
17
79
  /** A lookup from a result back to its stable rule handle, built from the loaded rules. */
18
80
  export function handleMap(rules) {
19
81
  const m = new Map();
@@ -21,11 +83,11 @@ export function handleMap(rules) {
21
83
  m.set(`${rule.source}:${rule.id}`, ruleFingerprint(rule));
22
84
  return (r) => m.get(`${r.ruleSource}:${r.ruleId}`) ?? "";
23
85
  }
24
- /** FAILs that are NOT configured as warnings — these fail the build. */
86
+ /** FAILs at `error` mode — these fail the build. (`off` never reaches here.) */
25
87
  export function blockingFailures(results, config, handleFor) {
26
- return results.filter((r) => r.status === "FAIL" && !config.warn.includes(handleFor(r)));
88
+ return results.filter((r) => r.status === "FAIL" && modeForResult(r, config, handleFor) === "error");
27
89
  }
28
- /** FAILs that ARE configured as warnings — shown, but they do not fail the build. */
90
+ /** FAILs at `warn` mode — shown, but they do not fail the build. */
29
91
  export function warningFailures(results, config, handleFor) {
30
- return results.filter((r) => r.status === "FAIL" && config.warn.includes(handleFor(r)));
92
+ return results.filter((r) => r.status === "FAIL" && modeForResult(r, config, handleFor) === "warn");
31
93
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "rulereceipt",
3
- "version": "0.1.48",
3
+ "version": "0.1.50",
4
4
  "description": "Checks whether a Claude Code session actually followed your CLAUDE.md / AGENTS.md rules, with evidence.",
5
5
  "repository": {
6
6
  "type": "git",