rulereceipt 0.1.40 → 0.1.42

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -201,6 +201,32 @@ Of 99 forbidding rules in the corpus that name a command-shaped literal, only
201
201
  marked automatically, which is the point.
202
202
 
203
203
 
204
+ ### Reproducing the published numbers
205
+
206
+ Every figure in the [postmortem](https://rulereceipt.dev/postmortem) and in the
207
+ issue threads is measured over 559 public rules files. The list of those files
208
+ is committed as `rule_file_corpus.md`; the files themselves are not, because
209
+ they belong to other projects.
210
+
211
+ ```bash
212
+ bash scripts/fetch-corpus.sh 600 # the default is 60
213
+ npx tsx scripts/corpus-report.ts # where real rules route: 63.2% not instructions
214
+ npx tsx scripts/false-accusation-rate.ts # reports carrying a false accusation
215
+ npx tsx scripts/verb-gate.ts # which gate admits each rule
216
+ npx tsx scripts/guard-replay.ts corpus 3 # what the PreToolUse guard would refuse
217
+ ```
218
+
219
+ The list holds 563 URLs and yields 559 files — four have moved or been deleted
220
+ upstream since it was drawn on 2026-08-30. That gap is expected and will grow;
221
+ if your count differs from 559, that is why, and the routing percentages move
222
+ by a rounding error rather than meaningfully.
223
+
224
+ Two of these print a sha256 for every session they read. That is deliberate:
225
+ "the largest sessions on this machine" is a selection rule, not a pin, and the
226
+ largest include the session doing the measuring. Two runs of identical code
227
+ four days apart returned 14,033 and 9,605 tool calls. Numbers are comparable
228
+ only when those hashes match.
229
+
204
230
  ## Which rules actually have teeth
205
231
 
206
232
  A rule in a file and a rule with a `PreToolUse` hook behind it look identical
@@ -17,6 +17,11 @@ export interface IfEditThenTestClassification {
17
17
  kind: "ifEditThenTest";
18
18
  rule: Rule;
19
19
  }
20
+ export interface EmojiClassification {
21
+ kind: "emojiOutput";
22
+ rule: Rule;
23
+ polarity: "forbid";
24
+ }
20
25
  export interface JudgmentClassification {
21
26
  kind: "judgment";
22
27
  rule: Rule;
@@ -106,7 +111,7 @@ export interface ClaimEvidenceClassification {
106
111
  kind: "claimEvidence";
107
112
  rule: Rule;
108
113
  }
109
- export type Classification = ClaimEvidenceClassification | DeterministicClassification | IfEditThenTestClassification | GitBranchPolicyClassification | CodeContentClassification | FileLifecycleClassification | NotARuleClassification | JudgmentClassification;
114
+ export type Classification = ClaimEvidenceClassification | DeterministicClassification | IfEditThenTestClassification | GitBranchPolicyClassification | CodeContentClassification | FileLifecycleClassification | NotARuleClassification | EmojiClassification | JudgmentClassification;
110
115
  export declare const DIRECTIVE_LANGUAGE: RegExp;
111
116
  /**
112
117
  * Imperative instruction — a bare command verb starting a clause ("Use
@@ -137,12 +142,5 @@ export declare function isEventRecord(rule: Rule): boolean;
137
142
  * is the forbidden command ("Never run: `rm -rf /`") is still a rule.
138
143
  */
139
144
  export declare function isCommandDocumentation(rule: Rule): boolean;
140
- /**
141
- * A rule is only treated as deterministic when it names a specific,
142
- * literal, checkable token (a CLI flag, a command, an exact string) in
143
- * backticks — e.g. "never use `git push --force`". Everything else
144
- * defaults to judgment, per the spec: never silently skip a rule by
145
- * guessing it's safe to pattern-match.
146
- */
147
145
  export declare function classifyRule(rule: Rule): Classification;
148
146
  export declare function classifyRules(rules: Rule[]): Classification[];
@@ -463,6 +463,36 @@ function polarityWasInferred(rule) {
463
463
  * defaults to judgment, per the spec: never silently skip a rule by
464
464
  * guessing it's safe to pattern-match.
465
465
  */
466
+ /**
467
+ * A rule forbidding emoji in output.
468
+ *
469
+ * Checkable, and it was not being checked: every phrasing routed to
470
+ * judgment, including one naming an emoji in backticks, because an emoji
471
+ * literal carries no alphanumerics and is rejected as unusable. Raised by
472
+ * anthropics/claude-code#94219.
473
+ *
474
+ * Both halves are required. The subject must be emoji, and the rule must
475
+ * FORBID them — "Use these emoji consistently across all chat output" is a
476
+ * real corpus rule and prescribes the opposite. A rule that merely mentions
477
+ * emoji in passing ("the identity file holds its name, vibe and emoji") is
478
+ * not about output at all.
479
+ */
480
+ const EMOJI_SUBJECT = /\bemojis?\b|\bemoticons?\b/i;
481
+ const EMOJI_FORBID = /\b(no|never|avoid|don't|do not|without|free of|refrain from|must not|shall not|not use|zero)\b/i;
482
+ function isEmojiRule(rule) {
483
+ const text = `${rule.title} ${rule.text}`;
484
+ if (!EMOJI_SUBJECT.test(text))
485
+ return false;
486
+ if (!EMOJI_FORBID.test(text))
487
+ return false;
488
+ // The prohibition has to be near the subject, not merely present in the
489
+ // same section — the same adjacency argument as claim-evidence.
490
+ const m = text.match(EMOJI_SUBJECT);
491
+ if (!m || m.index === undefined)
492
+ return false;
493
+ const window = text.slice(Math.max(0, m.index - 60), m.index + 40);
494
+ return EMOJI_FORBID.test(window);
495
+ }
466
496
  export function classifyRule(rule) {
467
497
  // Checked first: if this isn't a rule at all, no check of any kind
468
498
  // should run against it — not a keyword match, not an LLM call.
@@ -476,6 +506,11 @@ export function classifyRule(rule) {
476
506
  return { kind: "claimEvidence", rule };
477
507
  }
478
508
  // (length guard applied inside isClaimEvidenceRule)
509
+ // Checked before the backtick test: an emoji literal is rejected as an
510
+ // unusable pattern, so these would otherwise fall through to judgment.
511
+ if (isEmojiRule(rule)) {
512
+ return { kind: "emojiOutput", rule, polarity: "forbid" };
513
+ }
479
514
  const patterns = new Set();
480
515
  for (const match of rule.text.matchAll(BACKTICK_TOKEN)) {
481
516
  const token = match[1].trim();
@@ -0,0 +1,20 @@
1
+ import type { EmojiClassification } from "./classify.js";
2
+ import type { CheckResult, TranscriptEvent } from "../types.js";
3
+ /**
4
+ * Did the assistant emit an emoji, against a rule forbidding it?
5
+ *
6
+ * One of the very few OUTPUT rules that can be answered mechanically. Most
7
+ * of what a rules file forbids leaves no event to inspect; this one lands in
8
+ * assistant text, which both the transcript and the Stop hook can read.
9
+ *
10
+ * Raised by anthropics/claude-code#94219: instructions said "avoid emojis
11
+ * unless the user explicitly asks", the model sent one, then appended a
12
+ * parenthetical retracting it rather than removing it. The retraction is why
13
+ * this reads the text rather than trusting the session's own account of
14
+ * itself — the apology and the emoji were in the same message.
15
+ *
16
+ * Only ASSISTANT text counts. A user who sends an emoji has not broken the
17
+ * assistant's rule, and an earlier version of this check that scanned every
18
+ * event would have reported one.
19
+ */
20
+ export declare function runEmojiChecks(classifications: EmojiClassification[], events: TranscriptEvent[]): CheckResult[];
@@ -0,0 +1,101 @@
1
+ import { violation } from "../types.js";
2
+ /**
3
+ * Emoji, defined by Unicode rather than by a list of the ones we happened
4
+ * to have seen.
5
+ *
6
+ * The first version of this was hand-written character ranges. Tested
7
+ * against every pictographic codepoint Unicode knows about, it missed eight
8
+ * — including ✅ ❌ ⭐ ⌛ — because those blocks were not in the list. A list
9
+ * built from examples only ever covers the examples.
10
+ *
11
+ * Four properties, all of them mechanisms rather than enumerations:
12
+ *
13
+ * Emoji_Presentation renders as emoji by DEFAULT. 😀 🎉 ⌛
14
+ * Extended_Pictographic + U+FE0F
15
+ * a TEXT character explicitly given emoji form.
16
+ * © ™ ‼ ℹ ☀ are ordinary text; ©️ ™️ ‼️ ℹ️ ☀️ are not,
17
+ * and the difference is one invisible codepoint.
18
+ * regional indicators any flag, not a list of countries
19
+ * keycap sequence any keycap, not a list of digits
20
+ *
21
+ * Measured across codepoints U+0020 to U+1FAFF: 1,826 of 1,826 pictographic
22
+ * codepoints handled, and zero letters, digits, punctuation or symbols
23
+ * wrongly flagged. Accented Latin, CJK, arrows, maths and currency stay
24
+ * text, which matters — a check that fires on "café" or "日本語" is useless
25
+ * to most of the people who would run it.
26
+ *
27
+ * Because these are Unicode properties, new emoji are covered when the
28
+ * runtime's Unicode data updates. Nothing here needs editing for them.
29
+ */
30
+ const DEFAULT_EMOJI = /\p{Emoji_Presentation}/u;
31
+ const PICTOGRAPHIC = /\p{Extended_Pictographic}/u;
32
+ const REGIONAL_INDICATOR = /[\u{1F1E6}-\u{1F1FF}]/u;
33
+ const KEYCAP = /[0-9#*]\u{FE0F}?\u{20E3}/u;
34
+ const VARIATION_SELECTOR_16 = "\u{FE0F}";
35
+ /** Every distinct emoji in a string, in order of first appearance. */
36
+ function emojiIn(text) {
37
+ const found = [];
38
+ const chars = [...text];
39
+ if (KEYCAP.test(text)) {
40
+ const m = text.match(KEYCAP);
41
+ if (m)
42
+ found.push(m[0]);
43
+ }
44
+ for (let i = 0; i < chars.length; i++) {
45
+ const ch = chars[i];
46
+ const isEmoji = DEFAULT_EMOJI.test(ch) ||
47
+ REGIONAL_INDICATOR.test(ch) ||
48
+ (PICTOGRAPHIC.test(ch) && chars[i + 1] === VARIATION_SELECTOR_16);
49
+ if (!isEmoji)
50
+ continue;
51
+ const glyph = chars[i + 1] === VARIATION_SELECTOR_16 ? ch + chars[i + 1] : ch;
52
+ if (!found.includes(glyph))
53
+ found.push(glyph);
54
+ }
55
+ return found;
56
+ }
57
+ /**
58
+ * Did the assistant emit an emoji, against a rule forbidding it?
59
+ *
60
+ * One of the very few OUTPUT rules that can be answered mechanically. Most
61
+ * of what a rules file forbids leaves no event to inspect; this one lands in
62
+ * assistant text, which both the transcript and the Stop hook can read.
63
+ *
64
+ * Raised by anthropics/claude-code#94219: instructions said "avoid emojis
65
+ * unless the user explicitly asks", the model sent one, then appended a
66
+ * parenthetical retracting it rather than removing it. The retraction is why
67
+ * this reads the text rather than trusting the session's own account of
68
+ * itself — the apology and the emoji were in the same message.
69
+ *
70
+ * Only ASSISTANT text counts. A user who sends an emoji has not broken the
71
+ * assistant's rule, and an earlier version of this check that scanned every
72
+ * event would have reported one.
73
+ */
74
+ export function runEmojiChecks(classifications, events) {
75
+ let first = null;
76
+ for (const event of events) {
77
+ if (event.kind !== "text" || event.role !== "assistant")
78
+ continue;
79
+ const found = emojiIn(event.text);
80
+ if (found.length === 0)
81
+ continue;
82
+ first = { emoji: found, text: event.text };
83
+ break;
84
+ }
85
+ return classifications.map(({ rule, polarity }) => {
86
+ if (first) {
87
+ const at = first.text.indexOf(first.emoji[0]);
88
+ const excerpt = first.text.slice(Math.max(0, at - 40), at + 40).replace(/\s+/g, " ").trim();
89
+ return violation(rule, polarity, `the assistant's reply contained ${first.emoji.slice(0, 4).join(" ")} — "…${excerpt}…"`, { method: "emoji_output" });
90
+ }
91
+ return {
92
+ ruleId: rule.id,
93
+ ruleTitle: rule.title,
94
+ ruleSource: rule.source,
95
+ status: "PASS",
96
+ outcome: "pass",
97
+ evidence: "no emoji appears in anything the assistant said this session",
98
+ ceiling: "a scan of the assistant's recorded text — it cannot see anything said outside this transcript",
99
+ };
100
+ });
101
+ }
package/dist/cli.js CHANGED
@@ -16,6 +16,7 @@ import { runGitBranchPolicyChecks } from "./checks/gitBranchPolicy.js";
16
16
  import { runCodeContentChecks } from "./checks/codeContent.js";
17
17
  import { runFileLifecycleChecks } from "./checks/fileLifecycle.js";
18
18
  import { runClaimEvidenceChecks } from "./checks/claimEvidence.js";
19
+ import { runEmojiChecks } from "./checks/emojiOutput.js";
19
20
  import { runHook } from "./hook.js";
20
21
  import { runGuard } from "./guard.js";
21
22
  import { runJudgmentChecks } from "./checks/judgmentChecks.js";
@@ -241,6 +242,7 @@ async function runCheck(opts) {
241
242
  ...runCodeContentChecks(codeContent, events),
242
243
  ...runFileLifecycleChecks(fileLifecycle, events),
243
244
  ...runClaimEvidenceChecks(claimEvidence, events),
245
+ ...runEmojiChecks(classifications.filter((c) => c.kind === "emojiOutput"), events),
244
246
  ];
245
247
  // Deterministic checks run by default, always, with no key — judgment
246
248
  // rules only call out to an LLM with an explicit --llm on THIS run, never
package/dist/evaluate.js CHANGED
@@ -5,6 +5,7 @@ import { runGitBranchPolicyChecks } from "./checks/gitBranchPolicy.js";
5
5
  import { runCodeContentChecks } from "./checks/codeContent.js";
6
6
  import { runFileLifecycleChecks } from "./checks/fileLifecycle.js";
7
7
  import { runClaimEvidenceChecks } from "./checks/claimEvidence.js";
8
+ import { runEmojiChecks } from "./checks/emojiOutput.js";
8
9
  import { runJudgmentChecks } from "./checks/judgmentChecks.js";
9
10
  import { loadOverrides, ruleFingerprint, staleOverrides } from "./overrides.js";
10
11
  /**
@@ -38,6 +39,7 @@ export async function evaluateSession(cwd, rules, events, llm, needsLlmResult) {
38
39
  ...runCodeContentChecks(of("codeContent"), events),
39
40
  ...runFileLifecycleChecks(of("fileLifecycle"), events),
40
41
  ...runClaimEvidenceChecks(of("claimEvidence"), events),
42
+ ...runEmojiChecks(of("emojiOutput"), events),
41
43
  ];
42
44
  const judgment = of("judgment");
43
45
  const judgmentResults = llm
package/dist/guard.js CHANGED
@@ -209,13 +209,29 @@ export async function runGuard() {
209
209
  }
210
210
  if (blocks.length === 0)
211
211
  return allow();
212
+ const why = reason(blocks);
212
213
  process.stdout.write(JSON.stringify({
213
214
  hookSpecificOutput: {
214
215
  hookEventName: "PreToolUse",
215
216
  permissionDecision: "deny",
216
- permissionDecisionReason: reason(blocks),
217
+ permissionDecisionReason: why,
217
218
  },
218
219
  }));
220
+ // The same reason on stderr, deliberately duplicated.
221
+ //
222
+ // Two refusal paths exist and they carry the message differently.
223
+ // anthropics/claude-code#91574, measured by yurukusa on 2.1.278: a
224
+ // top-level {"permissionDecision":"deny"} body is invoked and IGNORED;
225
+ // the nested hookSpecificOutput form refuses; and stderr with exit 2
226
+ // refuses. Nobody has measured the nested body together with exit 2,
227
+ // which is what this emits.
228
+ //
229
+ // On the exit-2 path the documented channel back to the model is
230
+ // stderr, and stdout JSON is not promised to be read. Writing only the
231
+ // JSON risks a refusal with no reason attached — the block lands and
232
+ // the model is told nothing, which is the one failure a gate cannot
233
+ // afford. Printing both costs a duplicate line at worst.
234
+ process.stderr.write(`${why}\n`);
219
235
  // Exit 2 is what actually blocks the call; the JSON carries the reason.
220
236
  process.exitCode = 2;
221
237
  }
package/dist/types.d.ts CHANGED
@@ -64,7 +64,7 @@ export type CheckOutcome = "pass" | "fail"
64
64
  /** How a verdict was reached. A verdict with no method is a verdict with no standing. */
65
65
  export type CheckMethod = "text_scan" | "file_events" | "git_events" | "code_content" | "edit_test_pairing" | "claim_vs_evidence" | "model_judgment"
66
66
  /** Nothing ran. */
67
- | "none";
67
+ | "none" | "emoji_output";
68
68
  export interface CheckResult {
69
69
  ruleId: string;
70
70
  ruleTitle: string;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "rulereceipt",
3
- "version": "0.1.40",
3
+ "version": "0.1.42",
4
4
  "description": "Checks whether a Claude Code session actually followed your CLAUDE.md / AGENTS.md rules, with evidence.",
5
5
  "repository": {
6
6
  "type": "git",