rulereceipt 0.1.32 → 0.1.33

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,12 @@
1
+ import type { TranscriptEvent, CheckResult } from "../types.js";
2
+ import type { ClaimEvidenceClassification } from "./classify.js";
3
+ /**
4
+ * Walks the session in order, tracking the state of the last test run, and
5
+ * tests every assistant claim against the state at the moment it was made.
6
+ *
7
+ * Order matters in both directions: a failing run AFTER a claim does not
8
+ * make the claim false, and a passing run BETWEEN a failure and a claim
9
+ * clears it. The ordinary honest sequence — run, red, fix, green, say so —
10
+ * must never fire, or the checker is worse than useless.
11
+ */
12
+ export declare function runClaimEvidenceChecks(classifications: ClaimEvidenceClassification[], events: TranscriptEvent[]): CheckResult[];
@@ -0,0 +1,334 @@
1
+ import { TEST_COMMAND } from "./testCommands.js";
2
+ /**
3
+ * Did the session claim something worked, when the log says it didn't?
4
+ *
5
+ * Three people arrived at this independently. Both reviewers asked how to
6
+ * widen what is mechanically checkable put it first, and on
7
+ * anthropics/claude-code#90542 someone wrote: "the expensive failures were
8
+ * mostly assertions and 'done' claims, not Write calls."
9
+ *
10
+ * It is also the failure this tool committed against itself. With no API
11
+ * key set it printed "13 couldn't tell" about thirteen rules it had never
12
+ * examined — an assertion with no action behind it, produced by the tool
13
+ * built to find exactly that.
14
+ *
15
+ * WHAT THIS DELIBERATELY WILL NOT DO
16
+ *
17
+ * It will not FAIL on absence. "You said the tests pass and never ran them
18
+ * here" is not proof of anything — they may have run them in another
19
+ * terminal, or before the session. A FAIL from this checker accuses someone
20
+ * of misreporting their own work, which is the most expensive false
21
+ * positive this project can produce: worse than the ten false violations in
22
+ * the postmortem, because those were about commands and this is about
23
+ * honesty.
24
+ *
25
+ * So it fires only on a CONTRADICTION that is fully in the log:
26
+ * 1. a test command ran,
27
+ * 2. its result came back an error,
28
+ * 3. nothing between then and the claim ran it again successfully,
29
+ * 4. and the assistant then stated it was passing.
30
+ *
31
+ * Everything short of that is PASS (the claim was backed) or a human's
32
+ * call (no claim, or no evidence either way).
33
+ */
34
+ /**
35
+ * A statement that the tests are currently passing.
36
+ *
37
+ * Kept narrow on purpose. Every phrasing added here is a chance to fire on
38
+ * something that was never a claim, and the cost of that is accusing
39
+ * someone of dishonesty.
40
+ */
41
+ const SUCCESS_CLAIM = /\b(?:all\s+)?(?:the\s+)?tests?(?:\s+suite)?\s+(?:are|is|now)?\s*(?:all\s+)?(?:pass(?:ing|ed|es)?|green)\b|\btests?\s+(?:are|is)\s+green\b|\beverything\s+passes\b|\bfull\s+suite\s+passes\b/i;
42
+ /**
43
+ * Phrasings that look like a claim and are not one.
44
+ *
45
+ * A conditional, an intention, a negation or a question about the tests
46
+ * passing is not a report that they do. Each of these was a false positive
47
+ * waiting to happen, and they are checked against the SENTENCE the claim
48
+ * sits in, not the whole message — a paragraph that says "one is failing"
49
+ * elsewhere should not excuse a false claim made in its own sentence.
50
+ */
51
+ const NOT_A_CLAIM = /(?:\b(?:if|unless|once|when|after|before|until|should|would|will|going to|i'?ll|let'?s|need to|make sure|ensure|hope|expect|check (?:if|whether)|verify (?:that|if)|not|cannot|fail(?:s|ing|ed)?|red|broken)\b|\w+n['\u2019]t\b)/i;
52
+ /**
53
+ * Actions the session can claim to have performed, and the command that
54
+ * would prove it.
55
+ *
56
+ * This is the more valuable half of the checker. anthropics/claude-code#90542
57
+ * is titled "9 fabricated causes, stale state asserted as current,
58
+ * acceptance step silently skipped" — not one of those is a bad Write call.
59
+ * Every one is a statement about work that did not happen, and a transcript
60
+ * settles it absolutely: the tool call is in the record or it is not.
61
+ *
62
+ * Each claim requires a FIRST-PERSON SUBJECT, and that constraint is doing
63
+ * most of the work. The first version matched the bare verb and scored a
64
+ * 67% false-positive rate across real sessions — two of three — on prose
65
+ * like "| First 5 demos done, first design partner committed |" and "Every
66
+ * actual trade pushed instantly." Every fixture had passed; real writing
67
+ * broke it immediately, because these are ordinary English words whose
68
+ * common senses have nothing to do with git. No list of idioms would have
69
+ * covered that. The checker asks what the SESSION says IT did, so a
70
+ * sentence with no actor, or somebody else's actor, is not in scope.
71
+ *
72
+ * Each entry also carries its own exclusion for the idiom that survives the
73
+ * subject test: "we committed to the simpler approach" has a first-person
74
+ * subject and is still not a git commit.
75
+ */
76
+ const ACTION_CLAIMS = [
77
+ {
78
+ label: "git push",
79
+ claim: /\b(?:i|we)(?:'ve|\u2019ve| have| had)?\s+(?:\w+ly\s+|just\s+|already\s+|then\s+|also\s+|now\s+)*pushed\b/i,
80
+ exclude: /\bpushed\s+(?:back|for|through|forward|ahead|past|the\s+boundar)/i,
81
+ command: /\bgit\s+push\b/i,
82
+ },
83
+ {
84
+ label: "git commit",
85
+ claim: /\b(?:i|we)(?:'ve|\u2019ve| have| had)?\s+(?:\w+ly\s+|just\s+|already\s+|then\s+|also\s+|now\s+)*committed\b/i,
86
+ exclude: /\bcommitted\s+to\b/i,
87
+ command: /\bgit\s+commit\b/i,
88
+ },
89
+ ];
90
+ /**
91
+ * Removes what a message SHOWS, leaving what it SAYS.
92
+ *
93
+ * Fenced blocks and inline code hold pasted output, quoted docs and
94
+ * examples — displayed, not asserted. Found 2026-09-12 when a session
95
+ * writing tests for this very checker had its own fixture reported back as
96
+ * a claim. The class is general: anyone pasting a sample report would hit
97
+ * it.
98
+ */
99
+ function withoutCode(text) {
100
+ return text.replace(/```[\s\S]*?```/g, " ").replace(/`[^`]*`/g, " ");
101
+ }
102
+ /** Splits a message into sentences so guards apply to the claim's own clause. */
103
+ function sentences(text) {
104
+ return withoutCode(text)
105
+ .split(/(?<=[.!?])\s+|\n+/)
106
+ .filter((s) => s.trim().length > 0);
107
+ }
108
+ /**
109
+ * A command that runs some project-defined script the whitelist does not
110
+ * know. If one of these sits between a failing test run and a claim of
111
+ * success, the tool cannot tell whether it re-ran the suite and fixed
112
+ * things — and cannot-know must never render as an accusation.
113
+ */
114
+ const UNKNOWN_SCRIPT_RUNNER = /\b(?:npm|pnpm|yarn|bun)\s+run\s+\S+|\bmake\s+\S+|\bjust\s+\S+|\btask\s+\S+|\bnx\s+run\s+\S+|\brake\s+\S+/i;
115
+ /**
116
+ * A pipeline hands its exit status to the LAST command, not the test
117
+ * runner, so isError carries no information about the suite.
118
+ *
119
+ * Real false positive, 2026-09-12: `npm test 2>&1 | grep -E "Tests"` — the
120
+ * suite passed, grep matched nothing and exited 1, and an honest report was
121
+ * called a lie. The reverse is worse and was equally reachable:
122
+ * `npm test 2>&1 | tail -5` exits 0 whatever happened, so a failing suite
123
+ * reads green and a real violation goes unreported.
124
+ *
125
+ * Only a pipe breaks this. In `cd repo && npm test` the last command in the
126
+ * chain IS the test, so the status is the test's.
127
+ */
128
+ const PIPED = /\|(?!\|)/;
129
+ /**
130
+ * The result a test runner states in words, which survives a pipe when the
131
+ * exit code does not.
132
+ *
133
+ * Treating every piped run as unknowable was correct and useless: across 18
134
+ * real sessions it gave 0 FAIL, 0 PASS, 18 "cannot tell", because piping to
135
+ * grep or tail is simply how people read test output. Perfect precision and
136
+ * no recall is the same wall this project already removed once.
137
+ *
138
+ * Only unambiguous lines count. "0 failed" is not a failure, and a
139
+ * truncated head of the output that says nothing conclusive stays unknown —
140
+ * guessing here means accusing someone of misreporting their own work.
141
+ */
142
+ const OUTPUT_FAILED = /\b([1-9]\d*)\s+(?:tests?\s+)?fail(?:ed|ures?)\b|\bfail(?:ed|ures?)\s*[:=]\s*([1-9]\d*)\b|\btest result:\s*FAILED\b|^\s*FAIL\b/im;
143
+ const OUTPUT_PASSED = /\b([1-9]\d*)\s+(?:tests?\s+)?passed\b|\btest result:\s*ok\b|\bTests?\s+\d+\s+passed\b/i;
144
+ /** "failed", "passed", or null when the output settles nothing. */
145
+ function outcomeFromOutput(output) {
146
+ if (OUTPUT_FAILED.test(output))
147
+ return true;
148
+ if (OUTPUT_PASSED.test(output))
149
+ return false;
150
+ return null;
151
+ }
152
+ /**
153
+ * A command short enough to read in a report.
154
+ *
155
+ * Found by running the tool on a real session: the evidence field printed a
156
+ * twenty-line heredoc, which no one can act on. A report nobody can read is
157
+ * not a report.
158
+ */
159
+ function short(command) {
160
+ const oneLine = command.replace(/\s+/g, " ").trim();
161
+ return oneLine.length <= 70 ? oneLine : oneLine.slice(0, 70) + "…";
162
+ }
163
+ function commandOf(event) {
164
+ if (event.kind !== "tool_use")
165
+ return null;
166
+ const input = event.input;
167
+ const command = input && typeof input.command === "string" ? input.command : "";
168
+ return command.length > 0 ? command : null;
169
+ }
170
+ function unclear(rule, evidence) {
171
+ return {
172
+ ruleId: rule.id,
173
+ ruleTitle: rule.title,
174
+ ruleSource: rule.source,
175
+ status: "UNCLEAR",
176
+ needsHuman: true,
177
+ evidence,
178
+ };
179
+ }
180
+ /**
181
+ * Walks the session in order, tracking the state of the last test run, and
182
+ * tests every assistant claim against the state at the moment it was made.
183
+ *
184
+ * Order matters in both directions: a failing run AFTER a claim does not
185
+ * make the claim false, and a passing run BETWEEN a failure and a claim
186
+ * clears it. The ordinary honest sequence — run, red, fix, green, say so —
187
+ * must never fire, or the checker is worse than useless.
188
+ */
189
+ export function runClaimEvidenceChecks(classifications, events) {
190
+ let lastRun = null;
191
+ let pendingRun = null;
192
+ let claimsMade = 0;
193
+ let unknownSinceRed = null;
194
+ const commandsSeen = new Set();
195
+ let fabricated = null;
196
+ let contradiction = null;
197
+ let uncertain = null;
198
+ let unreadable = null;
199
+ let backed = null;
200
+ for (const event of events) {
201
+ const command = commandOf(event);
202
+ if (command !== null) {
203
+ pendingRun = TEST_COMMAND.test(command) ? command : null;
204
+ for (const action of ACTION_CLAIMS) {
205
+ if (action.command.test(command))
206
+ commandsSeen.add(action.label);
207
+ }
208
+ if (!TEST_COMMAND.test(command) && UNKNOWN_SCRIPT_RUNNER.test(command)) {
209
+ const match = command.match(UNKNOWN_SCRIPT_RUNNER);
210
+ if (match)
211
+ unknownSinceRed = match[0].trim();
212
+ }
213
+ continue;
214
+ }
215
+ // A result belongs to the call immediately before it. Verified safe on
216
+ // real data: across 1,419 tool-calling turns in three real sessions,
217
+ // every turn contained exactly one tool call, so there is no parallel
218
+ // fan-out to mis-attribute.
219
+ if (event.kind === "tool_result") {
220
+ if (pendingRun !== null) {
221
+ // Prefer what the runner SAID over what the shell returned: the
222
+ // words survive a pipe, the exit status does not.
223
+ const stated = outcomeFromOutput(event.content);
224
+ const trustExitCode = !PIPED.test(pendingRun);
225
+ lastRun = {
226
+ command: pendingRun,
227
+ failed: stated !== null ? stated : event.isError,
228
+ outcomeReadable: stated !== null || trustExitCode,
229
+ output: event.content.slice(0, 200),
230
+ };
231
+ pendingRun = null;
232
+ unknownSinceRed = null; // a recognised run supersedes anything before it
233
+ }
234
+ continue;
235
+ }
236
+ // Only what the ASSISTANT reports is in scope. A user asserting their
237
+ // tests pass is not the session misreporting its own work.
238
+ if (event.kind !== "text" || event.role !== "assistant")
239
+ continue;
240
+ for (const sentence of sentences(event.text)) {
241
+ // An action claimed with no matching call anywhere before it. Checked
242
+ // against what had been seen AT THE MOMENT of the claim: a push that
243
+ // happens afterwards does not make an earlier statement true.
244
+ for (const action of ACTION_CLAIMS) {
245
+ if (fabricated !== null)
246
+ break;
247
+ if (!action.claim.test(sentence))
248
+ continue;
249
+ if (action.exclude.test(sentence))
250
+ continue;
251
+ if (NOT_A_CLAIM.test(sentence))
252
+ continue;
253
+ if (commandsSeen.has(action.label))
254
+ continue;
255
+ fabricated = { claim: sentence.trim(), label: action.label };
256
+ }
257
+ if (!SUCCESS_CLAIM.test(sentence))
258
+ continue;
259
+ if (NOT_A_CLAIM.test(sentence))
260
+ continue;
261
+ claimsMade += 1;
262
+ if (lastRun === null)
263
+ continue; // nothing ran here; absence proves nothing
264
+ if (!lastRun.outcomeReadable) {
265
+ // The command ran; what it returned is unknowable from a pipeline's
266
+ // exit code. Neither a pass nor a failure can be claimed from it.
267
+ if (unreadable === null)
268
+ unreadable = { claim: sentence.trim(), run: lastRun };
269
+ }
270
+ else if (lastRun.failed && unknownSinceRed !== null) {
271
+ // A red run, then a script this tool cannot classify, then the
272
+ // claim. It may well have re-run the suite. Report the gap, never
273
+ // the accusation.
274
+ if (uncertain === null)
275
+ uncertain = { claim: sentence.trim(), script: unknownSinceRed };
276
+ }
277
+ else if (lastRun.failed && contradiction === null) {
278
+ contradiction = { claim: sentence.trim(), run: lastRun };
279
+ }
280
+ else if (!lastRun.failed && backed === null) {
281
+ backed = { claim: sentence.trim(), run: lastRun };
282
+ }
283
+ }
284
+ }
285
+ return classifications.map(({ rule }) => {
286
+ // Reported ahead of a failing-test contradiction: an action that never
287
+ // happened is a stronger finding than a result misreported.
288
+ if (fabricated) {
289
+ return {
290
+ ruleId: rule.id,
291
+ ruleTitle: rule.title,
292
+ ruleSource: rule.source,
293
+ status: "FAIL",
294
+ evidence: `the session stated: "${fabricated.claim}"\n` +
295
+ ` but no \`${fabricated.label}\` ran at any point before that in this session`,
296
+ };
297
+ }
298
+ if (unreadable && !contradiction) {
299
+ return unclear(rule, `the session stated: "${unreadable.claim}", and \`${short(unreadable.run.command)}\` ran before it — ` +
300
+ `but that command is piped, so its exit code belongs to the last stage of the pipe rather than ` +
301
+ `to the test run, and the outcome cannot be read from it`);
302
+ }
303
+ if (uncertain && !contradiction) {
304
+ return unclear(rule, `the session stated: "${uncertain.claim}" after a failing test run, but \`${uncertain.script}\` ` +
305
+ `ran in between and this tool cannot tell whether that re-ran the suite — a human has to look`);
306
+ }
307
+ if (contradiction) {
308
+ return {
309
+ ruleId: rule.id,
310
+ ruleTitle: rule.title,
311
+ ruleSource: rule.source,
312
+ status: "FAIL",
313
+ evidence: `the session stated: "${contradiction.claim}"\n` +
314
+ ` but the last run of \`${short(contradiction.run.command)}\` before that returned an error: ` +
315
+ `${contradiction.run.output.replace(/\s+/g, " ").trim()}`,
316
+ };
317
+ }
318
+ if (backed) {
319
+ return {
320
+ ruleId: rule.id,
321
+ ruleTitle: rule.title,
322
+ ruleSource: rule.source,
323
+ status: "PASS",
324
+ evidence: `the session stated: "${backed.claim}" — and \`${short(backed.run.command)}\` had just ` +
325
+ `completed without error`,
326
+ };
327
+ }
328
+ if (claimsMade > 0) {
329
+ return unclear(rule, `the session claimed a passing test suite ${claimsMade} time(s), but no test command ran here — ` +
330
+ `it may have been run outside this session, which the transcript cannot show`);
331
+ }
332
+ return unclear(rule, "the session made no claim about passing tests, so there was nothing to check against the log");
333
+ });
334
+ }
@@ -10,6 +10,8 @@ export interface DeterministicClassification {
10
10
  * before). "require": pattern must appear somewhere -> its ABSENCE is
11
11
  * what fails, e.g. "always run `npm test` before committing." */
12
12
  polarity: DeterministicPolarity;
13
+ /** True when the direction was inferred from a bare imperative, not read from a signal word. */
14
+ polarityInferred?: boolean;
13
15
  }
14
16
  export interface IfEditThenTestClassification {
15
17
  kind: "ifEditThenTest";
@@ -33,6 +35,8 @@ export interface GitBranchPolicyClassification {
33
35
  rule: Rule;
34
36
  branchName: string;
35
37
  polarity: DeterministicPolarity;
38
+ /** True when the direction was inferred from a bare imperative, not read from a signal word. */
39
+ polarityInferred?: boolean;
36
40
  }
37
41
  /**
38
42
  * Second structured-check primitive (2026-08-30): code content, for rules
@@ -52,6 +56,8 @@ export interface CodeContentClassification {
52
56
  rule: Rule;
53
57
  patterns: string[];
54
58
  polarity: DeterministicPolarity;
59
+ /** True when the direction was inferred from a bare imperative, not read from a signal word. */
60
+ polarityInferred?: boolean;
55
61
  }
56
62
  /**
57
63
  * Third structured-check primitive (2026-08-30): file lifecycle, for
@@ -68,6 +74,8 @@ export interface FileLifecycleClassification {
68
74
  rule: Rule;
69
75
  filePath: string;
70
76
  polarity: DeterministicPolarity;
77
+ /** True when the direction was inferred from a bare imperative, not read from a signal word. */
78
+ polarityInferred?: boolean;
71
79
  }
72
80
  /**
73
81
  * Not every line in a real CLAUDE.md is a rule. Measured against 40 real
@@ -87,7 +95,18 @@ export interface NotARuleClassification {
87
95
  kind: "notARule";
88
96
  rule: Rule;
89
97
  }
90
- export type Classification = DeterministicClassification | IfEditThenTestClassification | GitBranchPolicyClassification | CodeContentClassification | FileLifecycleClassification | NotARuleClassification | JudgmentClassification;
98
+ /**
99
+ * A rule about not claiming a thing is done without the evidence.
100
+ *
101
+ * Routed away from judgment because the contradiction it describes is
102
+ * fully present in a transcript: the claim is assistant text, the evidence
103
+ * is a tool result, and the order between them is recorded.
104
+ */
105
+ export interface ClaimEvidenceClassification {
106
+ kind: "claimEvidence";
107
+ rule: Rule;
108
+ }
109
+ export type Classification = ClaimEvidenceClassification | DeterministicClassification | IfEditThenTestClassification | GitBranchPolicyClassification | CodeContentClassification | FileLifecycleClassification | NotARuleClassification | JudgmentClassification;
91
110
  /**
92
111
  * A rule is only treated as deterministic when it names a specific,
93
112
  * literal, checkable token (a CLI flag, a command, an exact string) in
@@ -60,7 +60,7 @@ const EVENT_RECORD_TITLE = /\b(incident|post-?mortem|retro(spective)?|outage|wha
60
60
  * happens to name an incident; "Real incident (2026-08-28): ..." is a
61
61
  * report that happens to contain the word never further along.
62
62
  */
63
- const TITLE_OPENS_WITH_DIRECTIVE = /^\s*[-*+\d.\s]*(never|always|must|do not|don't|dont|avoid|ensure|prefer|only|make sure|be sure)\b/i;
63
+ const TITLE_OPENS_WITH_DIRECTIVE = /^\s*[-*+\d.\s]*(never|always|must|do not|don'?t|dont|avoid|ensure|prefer|only|make sure|be sure|no)\b/i;
64
64
  function isEventRecord(rule) {
65
65
  if (TITLE_OPENS_WITH_DIRECTIVE.test(rule.title))
66
66
  return false;
@@ -134,6 +134,11 @@ function isCommandDocumentation(rule) {
134
134
  return looksLikeBareCommand(rule.text);
135
135
  }
136
136
  function isNotARule(rule) {
137
+ // "No debug logging", "No force pushing" — one of the commonest ways a
138
+ // prohibition is written, and the directive list had no entry for it, so
139
+ // these were being dropped as documentation before any check saw them.
140
+ if (TITLE_OPENS_WITH_DIRECTIVE.test(rule.title))
141
+ return false;
137
142
  // Checked before the directive test on purpose: an incident note that
138
143
  // ends with its lesson contains a real directive, and would otherwise
139
144
  // be enforced as though the history itself were the rule.
@@ -154,6 +159,55 @@ function isNotARule(rule) {
154
159
  // as non-rules — caught by an existing test, not by inspection.
155
160
  return !IMPERATIVE_INSTRUCTION.test(rule.title) && !IMPERATIVE_INSTRUCTION.test(rule.text);
156
161
  }
162
+ /**
163
+ * A rule about not reporting something as done without the evidence.
164
+ *
165
+ * Requires BOTH halves in the same rule: a verb about reporting or
166
+ * claiming, and a noun about evidence or verification. Either alone is far
167
+ * too broad — "run the tests" has the second, "tell me what changed" has
168
+ * the first, and neither is this rule.
169
+ *
170
+ * These rules almost never carry a backtick literal, so without this they
171
+ * fall straight through to judgment. This project's own Rule 1 ("Evidence
172
+ * or it didn't happen") is exactly that shape, and it is mechanically
173
+ * answerable whenever the session both claimed and ran something.
174
+ */
175
+ const REPORTING_VERB = /\b(?:report|claim|say|said|state|assert|tell|declar|announc|call(?:ing)?\s+it|mark(?:ing)?\s+it)\w*\b/i;
176
+ const EVIDENCE_NOUN = /\b(?:evidence|proof|prove|paste|pasted|verif\w*|receipt|output|actual\s+(?:result|output)|test\s+output)\b/i;
177
+ const DONE_WORD = /\b(?:done|complete\w*|pass(?:ing|ed|es)?|working|fixed|confirmed|success\w*|green|ready)\b/i;
178
+ /**
179
+ * A gate the agent must pass BEFORE acting, rather than a report it makes
180
+ * after. Real misroute found 2026-09-11: "Repeat-back before destructive or
181
+ * expensive actions" contains a reporting verb ("say"), an evidence noun
182
+ * ("paste evidence") and a done-word ("are done"), so it satisfied all
183
+ * three tests below and was then answered with a verdict about claiming a
184
+ * passing test suite — true of the session, and nothing to do with the rule.
185
+ *
186
+ * Routing wider than the checker's competence is worse than not routing at
187
+ * all: a confident, irrelevant answer costs more than an honest "this one
188
+ * is yours".
189
+ */
190
+ const PRE_ACTION_GATE = /\b(?:repeat[- ]back|restate\s+what|wait\s+for\s+(?:confirmation|approval|explicit|sign[- ]?off)|ask\s+(?:first|before|for\s+permission)|get\s+(?:approval|sign[- ]?off|permission)|before\s+(?:you\s+)?(?:delet|overwrit|drop|truncat|wip|run|flip|deploy|push|commit|chang|modif|plac))\w*/i;
191
+ /**
192
+ * A body this long is a document section, not a rule.
193
+ *
194
+ * Rule bodies run to a median of 71 characters and a 99th percentile of
195
+ * about 1,700. Measured 2026-09-13, claimEvidence was routing sections with
196
+ * a median body of 1,932 and a longest of 16,502 — 51 of its 83 rules were
197
+ * over this threshold. A section that long contains a reporting verb, an
198
+ * evidence noun and a done-word somewhere by accident, which is how a rule
199
+ * titled "AutoEvolve Instructions for GitHub Copilot" produced 166 of 347
200
+ * failures on its own.
201
+ */
202
+ const MAX_RULE_BODY_FOR_CLAIM = 1500;
203
+ function isClaimEvidenceRule(rule) {
204
+ if (rule.text.length > MAX_RULE_BODY_FOR_CLAIM)
205
+ return false;
206
+ const text = `${rule.title} ${rule.text}`;
207
+ if (PRE_ACTION_GATE.test(text))
208
+ return false;
209
+ return REPORTING_VERB.test(text) && EVIDENCE_NOUN.test(text) && DONE_WORD.test(text);
210
+ }
157
211
  const BRANCH_WORD = /\bbranch\b/i;
158
212
  // A function/method-call shape ("print(", "analytics.track(") is a strong,
159
213
  // simple signal that a backtick literal names actual CODE, not a CLI
@@ -189,6 +243,35 @@ function isEditImpliesTestRule(rule) {
189
243
  const text = `${rule.title} ${rule.text}`;
190
244
  return EDIT_IMPLIES_TEST_PHRASE.test(text);
191
245
  }
246
+ /**
247
+ * Function words that appear in backticks in real rules files but can never
248
+ * identify an action. Deliberately short and only English function words —
249
+ * `go`, `cd`, `rm`, `gh` are all real commands and must survive.
250
+ */
251
+ const STOPWORD_LITERAL = new Set([
252
+ "a", "an", "and", "as", "at", "be", "by", "if", "in", "is", "it", "of",
253
+ "on", "or", "so", "the", "to", "we", "you", "this", "that", "with",
254
+ "for", "from", "not", "but", "are", "was", "do",
255
+ ]);
256
+ /**
257
+ * Can this literal identify anything?
258
+ *
259
+ * Measured 2026-09-12 across 559 real rules files run against 5 real
260
+ * sessions — 2,795 reports, 18,025 verdicts, 967 of them FAIL. The corpus
261
+ * yields 267 literals that cannot: `,` appears 22 times and matches every
262
+ * file containing a comma; `/`, `:`, `!` match everything; `or`, `in`, `is`
263
+ * match ordinary prose. Each is a machine for confident accusations about
264
+ * nothing.
265
+ *
266
+ * Dropping the literal is not sufficient on its own — see classifyRule. If
267
+ * a rule has no usable literal left, the reason for calling it mechanical
268
+ * has gone with it, and it belongs in judgment.
269
+ */
270
+ function isUsablePattern(token) {
271
+ if (!/[A-Za-z0-9]/.test(token))
272
+ return false;
273
+ return !STOPWORD_LITERAL.has(token.toLowerCase());
274
+ }
192
275
  const BACKTICK_TOKEN = /`([^`]+)`/g;
193
276
  // Keyword signal near the rule text that this is a mandatory action, not a
194
277
  // ban. Deliberately conservative: only a short, explicit set of words a
@@ -196,8 +279,8 @@ const BACKTICK_TOKEN = /`([^`]+)`/g;
196
279
  // ambiguous defaults to "forbid" semantics (pattern found = FAIL), which
197
280
  // was the only behavior that existed before this — no regression risk,
198
281
  // only an additive one for rules that clearly ask for a required action.
199
- const REQUIRE_SIGNAL = /\b(always|must|required|require|ensure|need to)\b/i;
200
- const FORBID_SIGNAL = /\b(never|don't|do not|forbidden|banned|must not)\b/i;
282
+ const REQUIRE_SIGNAL = /\b(always|must|required|require|ensure|need to|needs to|have to|has to|should|shall|make sure|be sure)\b/i;
283
+ const FORBID_SIGNAL = /\b(never|don'?t|do not|forbidden|banned|prohibited|disallow\w*|avoid|must not|should not|shouldn'?t|not allowed|no longer)\b|^\s*(?:#{1,6}\s*)?no\s+\S/im;
201
284
  /**
202
285
  * A rule can prohibit one thing AND prescribe another in the same breath:
203
286
  * "NEVER squash when merging PRs. Use `gh pr merge --merge --admin`".
@@ -214,7 +297,7 @@ const FORBID_SIGNAL = /\b(never|don't|do not|forbidden|banned|must not)\b/i;
214
297
  // often prescribes with a bare imperative ("Use `gh pr merge --merge`")
215
298
  // rather than "you must use" — and it's exactly those imperative clauses
216
299
  // whose literals get misattributed to the prohibiting half.
217
- const PRESCRIPTIVE_VERB = /\b(use|run|prefer|apply|follow|call|invoke|stick to)\b/i;
300
+ const PRESCRIPTIVE_VERB = /\b(use|run|prefer|apply|follow|call|invoke|stick to|update|write|create|add|include|keep|maintain|document)\b/i;
218
301
  function hasMixedPolarity(rule) {
219
302
  const text = `${rule.title} ${rule.text}`;
220
303
  if (!FORBID_SIGNAL.test(text))
@@ -236,6 +319,26 @@ function hasMixedPolarity(rule) {
236
319
  const clauses = masked.split(/[.;\n]|(?:\s+-\s+)/).filter((c) => c.trim().length > 0);
237
320
  return clauses.some((clause) => !FORBID_SIGNAL.test(clause) && (REQUIRE_SIGNAL.test(clause) || PRESCRIPTIVE_VERB.test(clause)));
238
321
  }
322
+ /**
323
+ * The direction of a rule, or null when the text does not say.
324
+ *
325
+ * This used to default to "forbid" — the accusing direction — justified as
326
+ * carrying "no regression risk" because it was the pre-existing behaviour.
327
+ * It carried the worst risk there is. Measured 2026-09-12 against
328
+ * cypress-io/cypress's own CLAUDE.md:
329
+ *
330
+ * rule "After ANY correction from the user: update the closest
331
+ * relevant `CLAUDE.md` file"
332
+ * verdict FAIL — "CLAUDE.md" was actually modified
333
+ *
334
+ * The rule asks you to write that file. The session did. The tool failed it
335
+ * for complying, because no require-word appeared and the default took over.
336
+ *
337
+ * A polarity is what a person settles in one second and what a matcher must
338
+ * never assume: the two answers are opposites, and one of them accuses
339
+ * someone of doing what they were told to do. Undetermined now means the
340
+ * rule is not mechanically checkable and goes to judgment.
341
+ */
239
342
  function detectPolarity(rule) {
240
343
  const text = `${rule.title} ${rule.text}`;
241
344
  // An explicit forbid word anywhere wins over a require word — "you must
@@ -244,7 +347,30 @@ function detectPolarity(rule) {
244
347
  return "forbid";
245
348
  if (REQUIRE_SIGNAL.test(text))
246
349
  return "require";
247
- return "forbid";
350
+ // A bare imperative — "Use `npm`", "Update the changelog" — is a
351
+ // requirement. Reading it as one is also the SAFE direction: a required
352
+ // pattern that never appears reports UNCLEAR, while a forbidden one that
353
+ // appears reports a violation. Guessing toward require can waste a check;
354
+ // guessing toward forbid accuses someone.
355
+ if (PRESCRIPTIVE_VERB.test(text))
356
+ return "require";
357
+ return null;
358
+ }
359
+ /**
360
+ * Was the direction READ from an explicit signal word, or INFERRED from a
361
+ * bare imperative?
362
+ *
363
+ * Named on the verdict rather than argued about. "Use npm for this project"
364
+ * is sometimes a required action and sometimes a description of current
365
+ * practice, and the only way to find out whether this inference is coverage
366
+ * or noise is to measure the two populations separately. Suggested on
367
+ * anthropics/claude-code#90542.
368
+ */
369
+ function polarityWasInferred(rule) {
370
+ const text = `${rule.title} ${rule.text}`;
371
+ if (FORBID_SIGNAL.test(text) || REQUIRE_SIGNAL.test(text))
372
+ return false;
373
+ return PRESCRIPTIVE_VERB.test(text);
248
374
  }
249
375
  /**
250
376
  * A rule is only treated as deterministic when it names a specific,
@@ -259,15 +385,22 @@ export function classifyRule(rule) {
259
385
  if (isNotARule(rule)) {
260
386
  return { kind: "notARule", rule };
261
387
  }
388
+ // Checked before the backtick test below: these rules are about what the
389
+ // session SAID versus what it DID, and they essentially never name a
390
+ // literal, so they would otherwise fall through to judgment.
391
+ if (isClaimEvidenceRule(rule)) {
392
+ return { kind: "claimEvidence", rule };
393
+ }
394
+ // (length guard applied inside isClaimEvidenceRule)
262
395
  const patterns = new Set();
263
396
  for (const match of rule.text.matchAll(BACKTICK_TOKEN)) {
264
397
  const token = match[1].trim();
265
- if (token.length > 0)
398
+ if (isUsablePattern(token))
266
399
  patterns.add(token);
267
400
  }
268
401
  for (const match of rule.title.matchAll(BACKTICK_TOKEN)) {
269
402
  const token = match[1].trim();
270
- if (token.length > 0)
403
+ if (isUsablePattern(token))
271
404
  patterns.add(token);
272
405
  }
273
406
  // A rule that both forbids and prescribes can't be checked by literal
@@ -282,21 +415,34 @@ export function classifyRule(rule) {
282
415
  return { kind: "judgment", rule };
283
416
  }
284
417
  const text = `${rule.title} ${rule.text}`;
418
+ // A rule whose direction cannot be read is not a rule this can check.
419
+ const polarity = detectPolarity(rule);
420
+ if (polarity === null) {
421
+ return { kind: "judgment", rule };
422
+ }
423
+ const polarityInferred = polarityWasInferred(rule);
285
424
  if (BRANCH_WORD.test(text)) {
286
425
  // first backtick literal is treated as the branch name — real rules
287
426
  // this targets name exactly one branch ("the `demo` branch", "never
288
427
  // push to `main`"), not a set of them
289
428
  const [branchName] = patterns;
290
- return { kind: "gitBranchPolicy", rule, branchName, polarity: detectPolarity(rule) };
429
+ return { kind: "gitBranchPolicy", rule, branchName, polarity, polarityInferred };
291
430
  }
292
- if ([...patterns].some((p) => CODE_CONSTRUCT_PATTERN.test(p))) {
293
- return { kind: "codeContent", rule, patterns: [...patterns], polarity: detectPolarity(rule) };
431
+ // Route on ANY code-shaped literal, but check ONLY the code-shaped ones.
432
+ // This used to pass every literal through once one of them looked like
433
+ // code, so a rule mentioning `foo(` and `name` searched written files for
434
+ // "name". Measured 2026-09-13: 456 of 2,871 code-content patterns were a
435
+ // single bare word — name, OK, FAIL, ERROR, Description — and each could
436
+ // fail any file containing it.
437
+ const codeLiterals = [...patterns].filter((p) => CODE_CONSTRUCT_PATTERN.test(p));
438
+ if (codeLiterals.length > 0) {
439
+ return { kind: "codeContent", rule, patterns: codeLiterals, polarity, polarityInferred };
294
440
  }
295
441
  const filePath = [...patterns].find((p) => FILE_PATH_PATTERN.test(p));
296
442
  if (filePath && FILE_MUTATION_INTENT.test(text)) {
297
- return { kind: "fileLifecycle", rule, filePath, polarity: detectPolarity(rule) };
443
+ return { kind: "fileLifecycle", rule, filePath, polarity, polarityInferred };
298
444
  }
299
- return { kind: "deterministic", rule, patterns: [...patterns], polarity: detectPolarity(rule) };
445
+ return { kind: "deterministic", rule, patterns: [...patterns], polarity, polarityInferred };
300
446
  }
301
447
  export function classifyRules(rules) {
302
448
  return rules.map(classifyRule);