rulereceipt 0.1.39 → 0.1.41

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -184,7 +184,7 @@ survives that by saying UNCLEAR. A gate cannot — so you mark it:
184
184
  rulereceipt rules --forbid <handle> --literal "git push --force"
185
185
  ```
186
186
 
187
- Handles come from `rulereceipt check --show-skipped`. The mark is stored
187
+ Handles come from `rulereceipt rules --handles`. The mark is stored
188
188
  against the rule's content hash, and the guard blocks on that literal and no
189
189
  other. Three things it deliberately will not do:
190
190
 
@@ -201,6 +201,32 @@ Of 99 forbidding rules in the corpus that name a command-shaped literal, only
201
201
  marked automatically, which is the point.
202
202
 
203
203
 
204
+ ### Reproducing the published numbers
205
+
206
+ Every figure in the [postmortem](https://rulereceipt.dev/postmortem) and in the
207
+ issue threads is measured over 559 public rules files. The list of those files
208
+ is committed as `rule_file_corpus.md`; the files themselves are not, because
209
+ they belong to other projects.
210
+
211
+ ```bash
212
+ bash scripts/fetch-corpus.sh 600 # the default is 60
213
+ npx tsx scripts/corpus-report.ts # where real rules route: 63.2% not instructions
214
+ npx tsx scripts/false-accusation-rate.ts # reports carrying a false accusation
215
+ npx tsx scripts/verb-gate.ts # which gate admits each rule
216
+ npx tsx scripts/guard-replay.ts corpus 3 # what the PreToolUse guard would refuse
217
+ ```
218
+
219
+ The list holds 563 URLs and yields 559 files — four have moved or been deleted
220
+ upstream since it was drawn on 2026-08-30. That gap is expected and will grow;
221
+ if your count differs from 559, that is why, and the routing percentages move
222
+ by a rounding error rather than meaningfully.
223
+
224
+ Two of these print a sha256 for every session they read. That is deliberate:
225
+ "the largest sessions on this machine" is a selection rule, not a pin, and the
226
+ largest include the session doing the measuring. Two runs of identical code
227
+ four days apart returned 14,033 and 9,605 tool calls. Numbers are comparable
228
+ only when those hashes match.
229
+
204
230
  ## Which rules actually have teeth
205
231
 
206
232
  A rule in a file and a rule with a `PreToolUse` hook behind it look identical
@@ -17,6 +17,11 @@ export interface IfEditThenTestClassification {
17
17
  kind: "ifEditThenTest";
18
18
  rule: Rule;
19
19
  }
20
+ export interface EmojiClassification {
21
+ kind: "emojiOutput";
22
+ rule: Rule;
23
+ polarity: "forbid";
24
+ }
20
25
  export interface JudgmentClassification {
21
26
  kind: "judgment";
22
27
  rule: Rule;
@@ -106,7 +111,7 @@ export interface ClaimEvidenceClassification {
106
111
  kind: "claimEvidence";
107
112
  rule: Rule;
108
113
  }
109
- export type Classification = ClaimEvidenceClassification | DeterministicClassification | IfEditThenTestClassification | GitBranchPolicyClassification | CodeContentClassification | FileLifecycleClassification | NotARuleClassification | JudgmentClassification;
114
+ export type Classification = ClaimEvidenceClassification | DeterministicClassification | IfEditThenTestClassification | GitBranchPolicyClassification | CodeContentClassification | FileLifecycleClassification | NotARuleClassification | EmojiClassification | JudgmentClassification;
110
115
  export declare const DIRECTIVE_LANGUAGE: RegExp;
111
116
  /**
112
117
  * Imperative instruction — a bare command verb starting a clause ("Use
@@ -137,12 +142,5 @@ export declare function isEventRecord(rule: Rule): boolean;
137
142
  * is the forbidden command ("Never run: `rm -rf /`") is still a rule.
138
143
  */
139
144
  export declare function isCommandDocumentation(rule: Rule): boolean;
140
- /**
141
- * A rule is only treated as deterministic when it names a specific,
142
- * literal, checkable token (a CLI flag, a command, an exact string) in
143
- * backticks — e.g. "never use `git push --force`". Everything else
144
- * defaults to judgment, per the spec: never silently skip a rule by
145
- * guessing it's safe to pattern-match.
146
- */
147
145
  export declare function classifyRule(rule: Rule): Classification;
148
146
  export declare function classifyRules(rules: Rule[]): Classification[];
@@ -463,6 +463,36 @@ function polarityWasInferred(rule) {
463
463
  * defaults to judgment, per the spec: never silently skip a rule by
464
464
  * guessing it's safe to pattern-match.
465
465
  */
466
+ /**
467
+ * A rule forbidding emoji in output.
468
+ *
469
+ * Checkable, and it was not being checked: every phrasing routed to
470
+ * judgment, including one naming an emoji in backticks, because an emoji
471
+ * literal carries no alphanumerics and is rejected as unusable. Raised by
472
+ * anthropics/claude-code#94219.
473
+ *
474
+ * Both halves are required. The subject must be emoji, and the rule must
475
+ * FORBID them — "Use these emoji consistently across all chat output" is a
476
+ * real corpus rule and prescribes the opposite. A rule that merely mentions
477
+ * emoji in passing ("the identity file holds its name, vibe and emoji") is
478
+ * not about output at all.
479
+ */
480
+ const EMOJI_SUBJECT = /\bemojis?\b|\bemoticons?\b/i;
481
+ const EMOJI_FORBID = /\b(no|never|avoid|don't|do not|without|free of|refrain from|must not|shall not|not use|zero)\b/i;
482
+ function isEmojiRule(rule) {
483
+ const text = `${rule.title} ${rule.text}`;
484
+ if (!EMOJI_SUBJECT.test(text))
485
+ return false;
486
+ if (!EMOJI_FORBID.test(text))
487
+ return false;
488
+ // The prohibition has to be near the subject, not merely present in the
489
+ // same section — the same adjacency argument as claim-evidence.
490
+ const m = text.match(EMOJI_SUBJECT);
491
+ if (!m || m.index === undefined)
492
+ return false;
493
+ const window = text.slice(Math.max(0, m.index - 60), m.index + 40);
494
+ return EMOJI_FORBID.test(window);
495
+ }
466
496
  export function classifyRule(rule) {
467
497
  // Checked first: if this isn't a rule at all, no check of any kind
468
498
  // should run against it — not a keyword match, not an LLM call.
@@ -476,6 +506,11 @@ export function classifyRule(rule) {
476
506
  return { kind: "claimEvidence", rule };
477
507
  }
478
508
  // (length guard applied inside isClaimEvidenceRule)
509
+ // Checked before the backtick test: an emoji literal is rejected as an
510
+ // unusable pattern, so these would otherwise fall through to judgment.
511
+ if (isEmojiRule(rule)) {
512
+ return { kind: "emojiOutput", rule, polarity: "forbid" };
513
+ }
479
514
  const patterns = new Set();
480
515
  for (const match of rule.text.matchAll(BACKTICK_TOKEN)) {
481
516
  const token = match[1].trim();
@@ -0,0 +1,20 @@
1
+ import type { EmojiClassification } from "./classify.js";
2
+ import type { CheckResult, TranscriptEvent } from "../types.js";
3
+ /**
4
+ * Did the assistant emit an emoji, against a rule forbidding it?
5
+ *
6
+ * One of the very few OUTPUT rules that can be answered mechanically. Most
7
+ * of what a rules file forbids leaves no event to inspect; this one lands in
8
+ * assistant text, which both the transcript and the Stop hook can read.
9
+ *
10
+ * Raised by anthropics/claude-code#94219: instructions said "avoid emojis
11
+ * unless the user explicitly asks", the model sent one, then appended a
12
+ * parenthetical retracting it rather than removing it. The retraction is why
13
+ * this reads the text rather than trusting the session's own account of
14
+ * itself — the apology and the emoji were in the same message.
15
+ *
16
+ * Only ASSISTANT text counts. A user who sends an emoji has not broken the
17
+ * assistant's rule, and an earlier version of this check that scanned every
18
+ * event would have reported one.
19
+ */
20
+ export declare function runEmojiChecks(classifications: EmojiClassification[], events: TranscriptEvent[]): CheckResult[];
@@ -0,0 +1,70 @@
1
+ import { violation } from "../types.js";
2
+ /**
3
+ * Emoji, as distinct from "any character a keyboard cannot type".
4
+ *
5
+ * Deliberately narrow. Accented Latin, CJK, mathematical symbols, arrows,
6
+ * dashes, degree signs and the check mark are NOT emoji, and a rule banning
7
+ * emoji must not fire on "café", "日本語", "±3°C" or "2×3". Those are
8
+ * ordinary text to the people who write them, and treating them as a
9
+ * violation would make this checker useless outside English.
10
+ *
11
+ * Covered: the pictographic blocks, the emoticon block, transport and map
12
+ * symbols, supplemental symbols, flags, and the dingbats that are actually
13
+ * rendered as emoji. Variation-selector-16 is included because it is what
14
+ * turns an otherwise plain glyph into its emoji presentation.
15
+ */
16
+ const EMOJI = /[\u{1F300}-\u{1F5FF}\u{1F600}-\u{1F64F}\u{1F680}-\u{1F6FF}\u{1F900}-\u{1FAFF}\u{1F1E6}-\u{1F1FF}\u{2600}-\u{26FF}\u{FE0F}]/u;
17
+ /** Every distinct emoji in a string, in order of first appearance. */
18
+ function emojiIn(text) {
19
+ const found = [];
20
+ for (const ch of text) {
21
+ if (EMOJI.test(ch) && ch !== "️" && !found.includes(ch))
22
+ found.push(ch);
23
+ }
24
+ return found;
25
+ }
26
+ /**
27
+ * Did the assistant emit an emoji, against a rule forbidding it?
28
+ *
29
+ * One of the very few OUTPUT rules that can be answered mechanically. Most
30
+ * of what a rules file forbids leaves no event to inspect; this one lands in
31
+ * assistant text, which both the transcript and the Stop hook can read.
32
+ *
33
+ * Raised by anthropics/claude-code#94219: instructions said "avoid emojis
34
+ * unless the user explicitly asks", the model sent one, then appended a
35
+ * parenthetical retracting it rather than removing it. The retraction is why
36
+ * this reads the text rather than trusting the session's own account of
37
+ * itself — the apology and the emoji were in the same message.
38
+ *
39
+ * Only ASSISTANT text counts. A user who sends an emoji has not broken the
40
+ * assistant's rule, and an earlier version of this check that scanned every
41
+ * event would have reported one.
42
+ */
43
+ export function runEmojiChecks(classifications, events) {
44
+ let first = null;
45
+ for (const event of events) {
46
+ if (event.kind !== "text" || event.role !== "assistant")
47
+ continue;
48
+ const found = emojiIn(event.text);
49
+ if (found.length === 0)
50
+ continue;
51
+ first = { emoji: found, text: event.text };
52
+ break;
53
+ }
54
+ return classifications.map(({ rule, polarity }) => {
55
+ if (first) {
56
+ const at = first.text.indexOf(first.emoji[0]);
57
+ const excerpt = first.text.slice(Math.max(0, at - 40), at + 40).replace(/\s+/g, " ").trim();
58
+ return violation(rule, polarity, `the assistant's reply contained ${first.emoji.slice(0, 4).join(" ")} — "…${excerpt}…"`, { method: "emoji_output" });
59
+ }
60
+ return {
61
+ ruleId: rule.id,
62
+ ruleTitle: rule.title,
63
+ ruleSource: rule.source,
64
+ status: "PASS",
65
+ outcome: "pass",
66
+ evidence: "no emoji appears in anything the assistant said this session",
67
+ ceiling: "a scan of the assistant's recorded text — it cannot see anything said outside this transcript",
68
+ };
69
+ });
70
+ }
package/dist/cli.js CHANGED
@@ -16,6 +16,7 @@ import { runGitBranchPolicyChecks } from "./checks/gitBranchPolicy.js";
16
16
  import { runCodeContentChecks } from "./checks/codeContent.js";
17
17
  import { runFileLifecycleChecks } from "./checks/fileLifecycle.js";
18
18
  import { runClaimEvidenceChecks } from "./checks/claimEvidence.js";
19
+ import { runEmojiChecks } from "./checks/emojiOutput.js";
19
20
  import { runHook } from "./hook.js";
20
21
  import { runGuard } from "./guard.js";
21
22
  import { runJudgmentChecks } from "./checks/judgmentChecks.js";
@@ -241,6 +242,7 @@ async function runCheck(opts) {
241
242
  ...runCodeContentChecks(codeContent, events),
242
243
  ...runFileLifecycleChecks(fileLifecycle, events),
243
244
  ...runClaimEvidenceChecks(claimEvidence, events),
245
+ ...runEmojiChecks(classifications.filter((c) => c.kind === "emojiOutput"), events),
244
246
  ];
245
247
  // Deterministic checks run by default, always, with no key — judgment
246
248
  // rules only call out to an LLM with an explicit --llm on THIS run, never
@@ -535,6 +537,33 @@ async function runRules(opts) {
535
537
  console.log(`your judgment onto words you never read.`);
536
538
  return;
537
539
  }
540
+ /**
541
+ * Every rule with its handle.
542
+ *
543
+ * --forbid needs a handle, and before this the only place handles were
544
+ * printed was `check --show-skipped`, which lists the items the classifier
545
+ * DISCARDED. A rule that is actually being checked had no handle anywhere,
546
+ * so the marking feature shipped in 0.1.39 could not be reached for any
547
+ * rule a user would want to mark. Found by trying to use it.
548
+ */
549
+ if (opts.handles) {
550
+ if (rules.length === 0) {
551
+ console.log("No CLAUDE.md or AGENTS.md rules found in this project.");
552
+ return;
553
+ }
554
+ console.log(`${rules.length} rule${rules.length === 1 ? "" : "s"} in this project:\n`);
555
+ for (const r of rules) {
556
+ const h = ruleFingerprint(r);
557
+ const marked = overrides.get(h)?.forbids;
558
+ console.log(` ${h} ${r.title.replace(/\s+/g, " ").trim().slice(0, 72)}`);
559
+ if (marked?.length)
560
+ console.log(` blocks on: ${marked.map((f) => `\`${f}\``).join(", ")}`);
561
+ }
562
+ console.log(`\nMark which clause of a rule is the prohibition:`);
563
+ console.log(` rulereceipt rules --forbid <handle> --literal "<the banned command>"`);
564
+ console.log(`Only a marked clause can ever refuse a command, and only via \`rulereceipt guard\`.`);
565
+ return;
566
+ }
538
567
  if (opts.clear) {
539
568
  console.log(clearOverride(cwd, opts.clear) ? `Removed the correction for ${opts.clear}.` : `No correction stored for ${opts.clear}.`);
540
569
  return;
@@ -651,6 +680,7 @@ program
651
680
  .description("correct what the classifier treats as a rule. Handles come from `check --show-skipped`.")
652
681
  .option("--include <handle>", "treat this item as a real rule and check it from now on")
653
682
  .option("--exclude <handle>", "treat this item as documentation and stop reporting it")
683
+ .option("--handles", "list every rule with its handle, for use with --forbid")
654
684
  .option("--forbid <handle>", "mark which clause of this rule is the prohibition, so the guard may block on it")
655
685
  .option("--literal <text>", "the exact banned command, used with --forbid; must appear in the rule")
656
686
  .option("--clear <handle>", "remove a stored correction")
package/dist/evaluate.js CHANGED
@@ -5,6 +5,7 @@ import { runGitBranchPolicyChecks } from "./checks/gitBranchPolicy.js";
5
5
  import { runCodeContentChecks } from "./checks/codeContent.js";
6
6
  import { runFileLifecycleChecks } from "./checks/fileLifecycle.js";
7
7
  import { runClaimEvidenceChecks } from "./checks/claimEvidence.js";
8
+ import { runEmojiChecks } from "./checks/emojiOutput.js";
8
9
  import { runJudgmentChecks } from "./checks/judgmentChecks.js";
9
10
  import { loadOverrides, ruleFingerprint, staleOverrides } from "./overrides.js";
10
11
  /**
@@ -38,6 +39,7 @@ export async function evaluateSession(cwd, rules, events, llm, needsLlmResult) {
38
39
  ...runCodeContentChecks(of("codeContent"), events),
39
40
  ...runFileLifecycleChecks(of("fileLifecycle"), events),
40
41
  ...runClaimEvidenceChecks(of("claimEvidence"), events),
42
+ ...runEmojiChecks(of("emojiOutput"), events),
41
43
  ];
42
44
  const judgment = of("judgment");
43
45
  const judgmentResults = llm
package/dist/guard.js CHANGED
@@ -80,13 +80,22 @@ function structuredBlocks(cwd, event) {
80
80
  function ratifiedLiteralBlocks(cwd, command) {
81
81
  const overrides = loadOverrides(cwd);
82
82
  const blocks = [];
83
- for (const c of forbidRules(cwd)) {
84
- if (c.kind !== "deterministic")
83
+ // Read from the RULES, not from the classification. A human mark
84
+ // supersedes the classifier, including its refusal to classify — and that
85
+ // refusal is the common case here. "Never use `git push --force`; prefer
86
+ // `git push --force-with-lease`" goes to judgment via hasMixedPolarity,
87
+ // for a correct reason: literal matching cannot tell which half owns which
88
+ // token. A rule naming both the ban and the alternative is the canonical
89
+ // reason to have someone say which is which, so gating the mark behind the
90
+ // classifier made the feature unavailable in exactly the case that
91
+ // motivated it. Shipped that way in 0.1.39 and caught by running it.
92
+ for (const rule of loadRules(cwd)) {
93
+ if (overrides.get(ruleFingerprint(rule))?.decision === "notARule")
85
94
  continue;
86
- for (const literal of ratifiedForbids(overrides, c.rule)) {
95
+ for (const literal of ratifiedForbids(overrides, rule)) {
87
96
  if (!commandRunsLiteral(command, literal))
88
97
  continue;
89
- blocks.push({ rule: c.rule, why: `the command about to run does \`${literal}\`, which this rule forbids (marked by you, not inferred)` });
98
+ blocks.push({ rule, why: `the command about to run does \`${literal}\`, which this rule forbids (marked by you, not inferred)` });
90
99
  break;
91
100
  }
92
101
  }
@@ -200,13 +209,29 @@ export async function runGuard() {
200
209
  }
201
210
  if (blocks.length === 0)
202
211
  return allow();
212
+ const why = reason(blocks);
203
213
  process.stdout.write(JSON.stringify({
204
214
  hookSpecificOutput: {
205
215
  hookEventName: "PreToolUse",
206
216
  permissionDecision: "deny",
207
- permissionDecisionReason: reason(blocks),
217
+ permissionDecisionReason: why,
208
218
  },
209
219
  }));
220
+ // The same reason on stderr, deliberately duplicated.
221
+ //
222
+ // Two refusal paths exist and they carry the message differently.
223
+ // anthropics/claude-code#91574, measured by yurukusa on 2.1.278: a
224
+ // top-level {"permissionDecision":"deny"} body is invoked and IGNORED;
225
+ // the nested hookSpecificOutput form refuses; and stderr with exit 2
226
+ // refuses. Nobody has measured the nested body together with exit 2,
227
+ // which is what this emits.
228
+ //
229
+ // On the exit-2 path the documented channel back to the model is
230
+ // stderr, and stdout JSON is not promised to be read. Writing only the
231
+ // JSON risks a refusal with no reason attached — the block lands and
232
+ // the model is told nothing, which is the one failure a gate cannot
233
+ // afford. Printing both costs a duplicate line at worst.
234
+ process.stderr.write(`${why}\n`);
210
235
  // Exit 2 is what actually blocks the call; the JSON carries the reason.
211
236
  process.exitCode = 2;
212
237
  }
package/dist/types.d.ts CHANGED
@@ -64,7 +64,7 @@ export type CheckOutcome = "pass" | "fail"
64
64
  /** How a verdict was reached. A verdict with no method is a verdict with no standing. */
65
65
  export type CheckMethod = "text_scan" | "file_events" | "git_events" | "code_content" | "edit_test_pairing" | "claim_vs_evidence" | "model_judgment"
66
66
  /** Nothing ran. */
67
- | "none";
67
+ | "none" | "emoji_output";
68
68
  export interface CheckResult {
69
69
  ruleId: string;
70
70
  ruleTitle: string;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "rulereceipt",
3
- "version": "0.1.39",
3
+ "version": "0.1.41",
4
4
  "description": "Checks whether a Claude Code session actually followed your CLAUDE.md / AGENTS.md rules, with evidence.",
5
5
  "repository": {
6
6
  "type": "git",