rulereceipt 0.1.32 → 0.1.34
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/checks/claimEvidence.d.ts +12 -0
- package/dist/checks/claimEvidence.js +334 -0
- package/dist/checks/classify.d.ts +20 -1
- package/dist/checks/classify.js +158 -12
- package/dist/checks/codeContent.js +13 -9
- package/dist/checks/deterministicChecks.js +19 -3
- package/dist/checks/fileLifecycle.js +35 -9
- package/dist/checks/gitBranchPolicy.js +28 -9
- package/dist/checks/ifEditThenTest.js +37 -1
- package/dist/checks/judgmentChecks.js +70 -8
- package/dist/checks/projectPaths.d.ts +16 -0
- package/dist/checks/projectPaths.js +18 -0
- package/dist/checks/testCommands.d.ts +23 -0
- package/dist/checks/testCommands.js +32 -0
- package/dist/cli.js +3 -0
- package/dist/report/generateReport.js +60 -11
- package/dist/types.d.ts +71 -0
- package/dist/types.js +22 -1
- package/package.json +2 -2
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
import type { TranscriptEvent, CheckResult } from "../types.js";
|
|
2
|
+
import type { ClaimEvidenceClassification } from "./classify.js";
|
|
3
|
+
/**
|
|
4
|
+
* Walks the session in order, tracking the state of the last test run, and
|
|
5
|
+
* tests every assistant claim against the state at the moment it was made.
|
|
6
|
+
*
|
|
7
|
+
* Order matters in both directions: a failing run AFTER a claim does not
|
|
8
|
+
* make the claim false, and a passing run BETWEEN a failure and a claim
|
|
9
|
+
* clears it. The ordinary honest sequence — run, red, fix, green, say so —
|
|
10
|
+
* must never fire, or the checker is worse than useless.
|
|
11
|
+
*/
|
|
12
|
+
export declare function runClaimEvidenceChecks(classifications: ClaimEvidenceClassification[], events: TranscriptEvent[]): CheckResult[];
|
|
@@ -0,0 +1,334 @@
|
|
|
1
|
+
import { TEST_COMMAND } from "./testCommands.js";
|
|
2
|
+
/**
|
|
3
|
+
* Did the session claim something worked, when the log says it didn't?
|
|
4
|
+
*
|
|
5
|
+
* Three people arrived at this independently. Both reviewers asked how to
|
|
6
|
+
* widen what is mechanically checkable put it first, and on
|
|
7
|
+
* anthropics/claude-code#90542 someone wrote: "the expensive failures were
|
|
8
|
+
* mostly assertions and 'done' claims, not Write calls."
|
|
9
|
+
*
|
|
10
|
+
* It is also the failure this tool committed against itself. With no API
|
|
11
|
+
* key set it printed "13 couldn't tell" about thirteen rules it had never
|
|
12
|
+
* examined — an assertion with no action behind it, produced by the tool
|
|
13
|
+
* built to find exactly that.
|
|
14
|
+
*
|
|
15
|
+
* WHAT THIS DELIBERATELY WILL NOT DO
|
|
16
|
+
*
|
|
17
|
+
* It will not FAIL on absence. "You said the tests pass and never ran them
|
|
18
|
+
* here" is not proof of anything — they may have run them in another
|
|
19
|
+
* terminal, or before the session. A FAIL from this checker accuses someone
|
|
20
|
+
* of misreporting their own work, which is the most expensive false
|
|
21
|
+
* positive this project can produce: worse than the ten false violations in
|
|
22
|
+
* the postmortem, because those were about commands and this is about
|
|
23
|
+
* honesty.
|
|
24
|
+
*
|
|
25
|
+
* So it fires only on a CONTRADICTION that is fully in the log:
|
|
26
|
+
* 1. a test command ran,
|
|
27
|
+
* 2. its result came back an error,
|
|
28
|
+
* 3. nothing between then and the claim ran it again successfully,
|
|
29
|
+
* 4. and the assistant then stated it was passing.
|
|
30
|
+
*
|
|
31
|
+
* Everything short of that is PASS (the claim was backed) or a human's
|
|
32
|
+
* call (no claim, or no evidence either way).
|
|
33
|
+
*/
|
|
34
|
+
/**
|
|
35
|
+
* A statement that the tests are currently passing.
|
|
36
|
+
*
|
|
37
|
+
* Kept narrow on purpose. Every phrasing added here is a chance to fire on
|
|
38
|
+
* something that was never a claim, and the cost of that is accusing
|
|
39
|
+
* someone of dishonesty.
|
|
40
|
+
*/
|
|
41
|
+
const SUCCESS_CLAIM = /\b(?:all\s+)?(?:the\s+)?tests?(?:\s+suite)?\s+(?:are|is|now)?\s*(?:all\s+)?(?:pass(?:ing|ed|es)?|green)\b|\btests?\s+(?:are|is)\s+green\b|\beverything\s+passes\b|\bfull\s+suite\s+passes\b/i;
|
|
42
|
+
/**
|
|
43
|
+
* Phrasings that look like a claim and are not one.
|
|
44
|
+
*
|
|
45
|
+
* A conditional, an intention, a negation or a question about the tests
|
|
46
|
+
* passing is not a report that they do. Each of these was a false positive
|
|
47
|
+
* waiting to happen, and they are checked against the SENTENCE the claim
|
|
48
|
+
* sits in, not the whole message — a paragraph that says "one is failing"
|
|
49
|
+
* elsewhere should not excuse a false claim made in its own sentence.
|
|
50
|
+
*/
|
|
51
|
+
const NOT_A_CLAIM = /(?:\b(?:if|unless|once|when|after|before|until|should|would|will|going to|i'?ll|let'?s|need to|make sure|ensure|hope|expect|check (?:if|whether)|verify (?:that|if)|not|cannot|fail(?:s|ing|ed)?|red|broken)\b|\w+n['\u2019]t\b)/i;
|
|
52
|
+
/**
|
|
53
|
+
* Actions the session can claim to have performed, and the command that
|
|
54
|
+
* would prove it.
|
|
55
|
+
*
|
|
56
|
+
* This is the more valuable half of the checker. anthropics/claude-code#90542
|
|
57
|
+
* is titled "9 fabricated causes, stale state asserted as current,
|
|
58
|
+
* acceptance step silently skipped" — not one of those is a bad Write call.
|
|
59
|
+
* Every one is a statement about work that did not happen, and a transcript
|
|
60
|
+
* settles it absolutely: the tool call is in the record or it is not.
|
|
61
|
+
*
|
|
62
|
+
* Each claim requires a FIRST-PERSON SUBJECT, and that constraint is doing
|
|
63
|
+
* most of the work. The first version matched the bare verb and scored a
|
|
64
|
+
* 67% false-positive rate across real sessions — two of three — on prose
|
|
65
|
+
* like "| First 5 demos done, first design partner committed |" and "Every
|
|
66
|
+
* actual trade pushed instantly." Every fixture had passed; real writing
|
|
67
|
+
* broke it immediately, because these are ordinary English words whose
|
|
68
|
+
* common senses have nothing to do with git. No list of idioms would have
|
|
69
|
+
* covered that. The checker asks what the SESSION says IT did, so a
|
|
70
|
+
* sentence with no actor, or somebody else's actor, is not in scope.
|
|
71
|
+
*
|
|
72
|
+
* Each entry also carries its own exclusion for the idiom that survives the
|
|
73
|
+
* subject test: "we committed to the simpler approach" has a first-person
|
|
74
|
+
* subject and is still not a git commit.
|
|
75
|
+
*/
|
|
76
|
+
const ACTION_CLAIMS = [
|
|
77
|
+
{
|
|
78
|
+
label: "git push",
|
|
79
|
+
claim: /\b(?:i|we)(?:'ve|\u2019ve| have| had)?\s+(?:\w+ly\s+|just\s+|already\s+|then\s+|also\s+|now\s+)*pushed\b/i,
|
|
80
|
+
exclude: /\bpushed\s+(?:back|for|through|forward|ahead|past|the\s+boundar)/i,
|
|
81
|
+
command: /\bgit\s+push\b/i,
|
|
82
|
+
},
|
|
83
|
+
{
|
|
84
|
+
label: "git commit",
|
|
85
|
+
claim: /\b(?:i|we)(?:'ve|\u2019ve| have| had)?\s+(?:\w+ly\s+|just\s+|already\s+|then\s+|also\s+|now\s+)*committed\b/i,
|
|
86
|
+
exclude: /\bcommitted\s+to\b/i,
|
|
87
|
+
command: /\bgit\s+commit\b/i,
|
|
88
|
+
},
|
|
89
|
+
];
|
|
90
|
+
/**
|
|
91
|
+
* Removes what a message SHOWS, leaving what it SAYS.
|
|
92
|
+
*
|
|
93
|
+
* Fenced blocks and inline code hold pasted output, quoted docs and
|
|
94
|
+
* examples — displayed, not asserted. Found 2026-09-12 when a session
|
|
95
|
+
* writing tests for this very checker had its own fixture reported back as
|
|
96
|
+
* a claim. The class is general: anyone pasting a sample report would hit
|
|
97
|
+
* it.
|
|
98
|
+
*/
|
|
99
|
+
function withoutCode(text) {
|
|
100
|
+
return text.replace(/```[\s\S]*?```/g, " ").replace(/`[^`]*`/g, " ");
|
|
101
|
+
}
|
|
102
|
+
/** Splits a message into sentences so guards apply to the claim's own clause. */
|
|
103
|
+
function sentences(text) {
|
|
104
|
+
return withoutCode(text)
|
|
105
|
+
.split(/(?<=[.!?])\s+|\n+/)
|
|
106
|
+
.filter((s) => s.trim().length > 0);
|
|
107
|
+
}
|
|
108
|
+
/**
|
|
109
|
+
* A command that runs some project-defined script the whitelist does not
|
|
110
|
+
* know. If one of these sits between a failing test run and a claim of
|
|
111
|
+
* success, the tool cannot tell whether it re-ran the suite and fixed
|
|
112
|
+
* things — and cannot-know must never render as an accusation.
|
|
113
|
+
*/
|
|
114
|
+
const UNKNOWN_SCRIPT_RUNNER = /\b(?:npm|pnpm|yarn|bun)\s+run\s+\S+|\bmake\s+\S+|\bjust\s+\S+|\btask\s+\S+|\bnx\s+run\s+\S+|\brake\s+\S+/i;
|
|
115
|
+
/**
|
|
116
|
+
* A pipeline hands its exit status to the LAST command, not the test
|
|
117
|
+
* runner, so isError carries no information about the suite.
|
|
118
|
+
*
|
|
119
|
+
* Real false positive, 2026-09-12: `npm test 2>&1 | grep -E "Tests"` — the
|
|
120
|
+
* suite passed, grep matched nothing and exited 1, and an honest report was
|
|
121
|
+
* called a lie. The reverse is worse and was equally reachable:
|
|
122
|
+
* `npm test 2>&1 | tail -5` exits 0 whatever happened, so a failing suite
|
|
123
|
+
* reads green and a real violation goes unreported.
|
|
124
|
+
*
|
|
125
|
+
* Only a pipe breaks this. In `cd repo && npm test` the last command in the
|
|
126
|
+
* chain IS the test, so the status is the test's.
|
|
127
|
+
*/
|
|
128
|
+
const PIPED = /\|(?!\|)/;
|
|
129
|
+
/**
|
|
130
|
+
* The result a test runner states in words, which survives a pipe when the
|
|
131
|
+
* exit code does not.
|
|
132
|
+
*
|
|
133
|
+
* Treating every piped run as unknowable was correct and useless: across 18
|
|
134
|
+
* real sessions it gave 0 FAIL, 0 PASS, 18 "cannot tell", because piping to
|
|
135
|
+
* grep or tail is simply how people read test output. Perfect precision and
|
|
136
|
+
* no recall is the same wall this project already removed once.
|
|
137
|
+
*
|
|
138
|
+
* Only unambiguous lines count. "0 failed" is not a failure, and a
|
|
139
|
+
* truncated head of the output that says nothing conclusive stays unknown —
|
|
140
|
+
* guessing here means accusing someone of misreporting their own work.
|
|
141
|
+
*/
|
|
142
|
+
const OUTPUT_FAILED = /\b([1-9]\d*)\s+(?:tests?\s+)?fail(?:ed|ures?)\b|\bfail(?:ed|ures?)\s*[:=]\s*([1-9]\d*)\b|\btest result:\s*FAILED\b|^\s*FAIL\b/im;
|
|
143
|
+
const OUTPUT_PASSED = /\b([1-9]\d*)\s+(?:tests?\s+)?passed\b|\btest result:\s*ok\b|\bTests?\s+\d+\s+passed\b/i;
|
|
144
|
+
/** "failed", "passed", or null when the output settles nothing. */
|
|
145
|
+
function outcomeFromOutput(output) {
|
|
146
|
+
if (OUTPUT_FAILED.test(output))
|
|
147
|
+
return true;
|
|
148
|
+
if (OUTPUT_PASSED.test(output))
|
|
149
|
+
return false;
|
|
150
|
+
return null;
|
|
151
|
+
}
|
|
152
|
+
/**
|
|
153
|
+
* A command short enough to read in a report.
|
|
154
|
+
*
|
|
155
|
+
* Found by running the tool on a real session: the evidence field printed a
|
|
156
|
+
* twenty-line heredoc, which no one can act on. A report nobody can read is
|
|
157
|
+
* not a report.
|
|
158
|
+
*/
|
|
159
|
+
function short(command) {
|
|
160
|
+
const oneLine = command.replace(/\s+/g, " ").trim();
|
|
161
|
+
return oneLine.length <= 70 ? oneLine : oneLine.slice(0, 70) + "…";
|
|
162
|
+
}
|
|
163
|
+
function commandOf(event) {
|
|
164
|
+
if (event.kind !== "tool_use")
|
|
165
|
+
return null;
|
|
166
|
+
const input = event.input;
|
|
167
|
+
const command = input && typeof input.command === "string" ? input.command : "";
|
|
168
|
+
return command.length > 0 ? command : null;
|
|
169
|
+
}
|
|
170
|
+
function unclear(rule, evidence) {
|
|
171
|
+
return {
|
|
172
|
+
ruleId: rule.id,
|
|
173
|
+
ruleTitle: rule.title,
|
|
174
|
+
ruleSource: rule.source,
|
|
175
|
+
status: "UNCLEAR",
|
|
176
|
+
needsHuman: true,
|
|
177
|
+
evidence,
|
|
178
|
+
};
|
|
179
|
+
}
|
|
180
|
+
/**
|
|
181
|
+
* Walks the session in order, tracking the state of the last test run, and
|
|
182
|
+
* tests every assistant claim against the state at the moment it was made.
|
|
183
|
+
*
|
|
184
|
+
* Order matters in both directions: a failing run AFTER a claim does not
|
|
185
|
+
* make the claim false, and a passing run BETWEEN a failure and a claim
|
|
186
|
+
* clears it. The ordinary honest sequence — run, red, fix, green, say so —
|
|
187
|
+
* must never fire, or the checker is worse than useless.
|
|
188
|
+
*/
|
|
189
|
+
export function runClaimEvidenceChecks(classifications, events) {
|
|
190
|
+
let lastRun = null;
|
|
191
|
+
let pendingRun = null;
|
|
192
|
+
let claimsMade = 0;
|
|
193
|
+
let unknownSinceRed = null;
|
|
194
|
+
const commandsSeen = new Set();
|
|
195
|
+
let fabricated = null;
|
|
196
|
+
let contradiction = null;
|
|
197
|
+
let uncertain = null;
|
|
198
|
+
let unreadable = null;
|
|
199
|
+
let backed = null;
|
|
200
|
+
for (const event of events) {
|
|
201
|
+
const command = commandOf(event);
|
|
202
|
+
if (command !== null) {
|
|
203
|
+
pendingRun = TEST_COMMAND.test(command) ? command : null;
|
|
204
|
+
for (const action of ACTION_CLAIMS) {
|
|
205
|
+
if (action.command.test(command))
|
|
206
|
+
commandsSeen.add(action.label);
|
|
207
|
+
}
|
|
208
|
+
if (!TEST_COMMAND.test(command) && UNKNOWN_SCRIPT_RUNNER.test(command)) {
|
|
209
|
+
const match = command.match(UNKNOWN_SCRIPT_RUNNER);
|
|
210
|
+
if (match)
|
|
211
|
+
unknownSinceRed = match[0].trim();
|
|
212
|
+
}
|
|
213
|
+
continue;
|
|
214
|
+
}
|
|
215
|
+
// A result belongs to the call immediately before it. Verified safe on
|
|
216
|
+
// real data: across 1,419 tool-calling turns in three real sessions,
|
|
217
|
+
// every turn contained exactly one tool call, so there is no parallel
|
|
218
|
+
// fan-out to mis-attribute.
|
|
219
|
+
if (event.kind === "tool_result") {
|
|
220
|
+
if (pendingRun !== null) {
|
|
221
|
+
// Prefer what the runner SAID over what the shell returned: the
|
|
222
|
+
// words survive a pipe, the exit status does not.
|
|
223
|
+
const stated = outcomeFromOutput(event.content);
|
|
224
|
+
const trustExitCode = !PIPED.test(pendingRun);
|
|
225
|
+
lastRun = {
|
|
226
|
+
command: pendingRun,
|
|
227
|
+
failed: stated !== null ? stated : event.isError,
|
|
228
|
+
outcomeReadable: stated !== null || trustExitCode,
|
|
229
|
+
output: event.content.slice(0, 200),
|
|
230
|
+
};
|
|
231
|
+
pendingRun = null;
|
|
232
|
+
unknownSinceRed = null; // a recognised run supersedes anything before it
|
|
233
|
+
}
|
|
234
|
+
continue;
|
|
235
|
+
}
|
|
236
|
+
// Only what the ASSISTANT reports is in scope. A user asserting their
|
|
237
|
+
// tests pass is not the session misreporting its own work.
|
|
238
|
+
if (event.kind !== "text" || event.role !== "assistant")
|
|
239
|
+
continue;
|
|
240
|
+
for (const sentence of sentences(event.text)) {
|
|
241
|
+
// An action claimed with no matching call anywhere before it. Checked
|
|
242
|
+
// against what had been seen AT THE MOMENT of the claim: a push that
|
|
243
|
+
// happens afterwards does not make an earlier statement true.
|
|
244
|
+
for (const action of ACTION_CLAIMS) {
|
|
245
|
+
if (fabricated !== null)
|
|
246
|
+
break;
|
|
247
|
+
if (!action.claim.test(sentence))
|
|
248
|
+
continue;
|
|
249
|
+
if (action.exclude.test(sentence))
|
|
250
|
+
continue;
|
|
251
|
+
if (NOT_A_CLAIM.test(sentence))
|
|
252
|
+
continue;
|
|
253
|
+
if (commandsSeen.has(action.label))
|
|
254
|
+
continue;
|
|
255
|
+
fabricated = { claim: sentence.trim(), label: action.label };
|
|
256
|
+
}
|
|
257
|
+
if (!SUCCESS_CLAIM.test(sentence))
|
|
258
|
+
continue;
|
|
259
|
+
if (NOT_A_CLAIM.test(sentence))
|
|
260
|
+
continue;
|
|
261
|
+
claimsMade += 1;
|
|
262
|
+
if (lastRun === null)
|
|
263
|
+
continue; // nothing ran here; absence proves nothing
|
|
264
|
+
if (!lastRun.outcomeReadable) {
|
|
265
|
+
// The command ran; what it returned is unknowable from a pipeline's
|
|
266
|
+
// exit code. Neither a pass nor a failure can be claimed from it.
|
|
267
|
+
if (unreadable === null)
|
|
268
|
+
unreadable = { claim: sentence.trim(), run: lastRun };
|
|
269
|
+
}
|
|
270
|
+
else if (lastRun.failed && unknownSinceRed !== null) {
|
|
271
|
+
// A red run, then a script this tool cannot classify, then the
|
|
272
|
+
// claim. It may well have re-run the suite. Report the gap, never
|
|
273
|
+
// the accusation.
|
|
274
|
+
if (uncertain === null)
|
|
275
|
+
uncertain = { claim: sentence.trim(), script: unknownSinceRed };
|
|
276
|
+
}
|
|
277
|
+
else if (lastRun.failed && contradiction === null) {
|
|
278
|
+
contradiction = { claim: sentence.trim(), run: lastRun };
|
|
279
|
+
}
|
|
280
|
+
else if (!lastRun.failed && backed === null) {
|
|
281
|
+
backed = { claim: sentence.trim(), run: lastRun };
|
|
282
|
+
}
|
|
283
|
+
}
|
|
284
|
+
}
|
|
285
|
+
return classifications.map(({ rule }) => {
|
|
286
|
+
// Reported ahead of a failing-test contradiction: an action that never
|
|
287
|
+
// happened is a stronger finding than a result misreported.
|
|
288
|
+
if (fabricated) {
|
|
289
|
+
return {
|
|
290
|
+
ruleId: rule.id,
|
|
291
|
+
ruleTitle: rule.title,
|
|
292
|
+
ruleSource: rule.source,
|
|
293
|
+
status: "FAIL",
|
|
294
|
+
evidence: `the session stated: "${fabricated.claim}"\n` +
|
|
295
|
+
` but no \`${fabricated.label}\` ran at any point before that in this session`,
|
|
296
|
+
};
|
|
297
|
+
}
|
|
298
|
+
if (unreadable && !contradiction) {
|
|
299
|
+
return unclear(rule, `the session stated: "${unreadable.claim}", and \`${short(unreadable.run.command)}\` ran before it — ` +
|
|
300
|
+
`but that command is piped, so its exit code belongs to the last stage of the pipe rather than ` +
|
|
301
|
+
`to the test run, and the outcome cannot be read from it`);
|
|
302
|
+
}
|
|
303
|
+
if (uncertain && !contradiction) {
|
|
304
|
+
return unclear(rule, `the session stated: "${uncertain.claim}" after a failing test run, but \`${uncertain.script}\` ` +
|
|
305
|
+
`ran in between and this tool cannot tell whether that re-ran the suite — a human has to look`);
|
|
306
|
+
}
|
|
307
|
+
if (contradiction) {
|
|
308
|
+
return {
|
|
309
|
+
ruleId: rule.id,
|
|
310
|
+
ruleTitle: rule.title,
|
|
311
|
+
ruleSource: rule.source,
|
|
312
|
+
status: "FAIL",
|
|
313
|
+
evidence: `the session stated: "${contradiction.claim}"\n` +
|
|
314
|
+
` but the last run of \`${short(contradiction.run.command)}\` before that returned an error: ` +
|
|
315
|
+
`${contradiction.run.output.replace(/\s+/g, " ").trim()}`,
|
|
316
|
+
};
|
|
317
|
+
}
|
|
318
|
+
if (backed) {
|
|
319
|
+
return {
|
|
320
|
+
ruleId: rule.id,
|
|
321
|
+
ruleTitle: rule.title,
|
|
322
|
+
ruleSource: rule.source,
|
|
323
|
+
status: "PASS",
|
|
324
|
+
evidence: `the session stated: "${backed.claim}" — and \`${short(backed.run.command)}\` had just ` +
|
|
325
|
+
`completed without error`,
|
|
326
|
+
};
|
|
327
|
+
}
|
|
328
|
+
if (claimsMade > 0) {
|
|
329
|
+
return unclear(rule, `the session claimed a passing test suite ${claimsMade} time(s), but no test command ran here — ` +
|
|
330
|
+
`it may have been run outside this session, which the transcript cannot show`);
|
|
331
|
+
}
|
|
332
|
+
return unclear(rule, "the session made no claim about passing tests, so there was nothing to check against the log");
|
|
333
|
+
});
|
|
334
|
+
}
|
|
@@ -10,6 +10,8 @@ export interface DeterministicClassification {
|
|
|
10
10
|
* before). "require": pattern must appear somewhere -> its ABSENCE is
|
|
11
11
|
* what fails, e.g. "always run `npm test` before committing." */
|
|
12
12
|
polarity: DeterministicPolarity;
|
|
13
|
+
/** True when the direction was inferred from a bare imperative, not read from a signal word. */
|
|
14
|
+
polarityInferred?: boolean;
|
|
13
15
|
}
|
|
14
16
|
export interface IfEditThenTestClassification {
|
|
15
17
|
kind: "ifEditThenTest";
|
|
@@ -33,6 +35,8 @@ export interface GitBranchPolicyClassification {
|
|
|
33
35
|
rule: Rule;
|
|
34
36
|
branchName: string;
|
|
35
37
|
polarity: DeterministicPolarity;
|
|
38
|
+
/** True when the direction was inferred from a bare imperative, not read from a signal word. */
|
|
39
|
+
polarityInferred?: boolean;
|
|
36
40
|
}
|
|
37
41
|
/**
|
|
38
42
|
* Second structured-check primitive (2026-08-30): code content, for rules
|
|
@@ -52,6 +56,8 @@ export interface CodeContentClassification {
|
|
|
52
56
|
rule: Rule;
|
|
53
57
|
patterns: string[];
|
|
54
58
|
polarity: DeterministicPolarity;
|
|
59
|
+
/** True when the direction was inferred from a bare imperative, not read from a signal word. */
|
|
60
|
+
polarityInferred?: boolean;
|
|
55
61
|
}
|
|
56
62
|
/**
|
|
57
63
|
* Third structured-check primitive (2026-08-30): file lifecycle, for
|
|
@@ -68,6 +74,8 @@ export interface FileLifecycleClassification {
|
|
|
68
74
|
rule: Rule;
|
|
69
75
|
filePath: string;
|
|
70
76
|
polarity: DeterministicPolarity;
|
|
77
|
+
/** True when the direction was inferred from a bare imperative, not read from a signal word. */
|
|
78
|
+
polarityInferred?: boolean;
|
|
71
79
|
}
|
|
72
80
|
/**
|
|
73
81
|
* Not every line in a real CLAUDE.md is a rule. Measured against 40 real
|
|
@@ -87,7 +95,18 @@ export interface NotARuleClassification {
|
|
|
87
95
|
kind: "notARule";
|
|
88
96
|
rule: Rule;
|
|
89
97
|
}
|
|
90
|
-
|
|
98
|
+
/**
|
|
99
|
+
* A rule about not claiming a thing is done without the evidence.
|
|
100
|
+
*
|
|
101
|
+
* Routed away from judgment because the contradiction it describes is
|
|
102
|
+
* fully present in a transcript: the claim is assistant text, the evidence
|
|
103
|
+
* is a tool result, and the order between them is recorded.
|
|
104
|
+
*/
|
|
105
|
+
export interface ClaimEvidenceClassification {
|
|
106
|
+
kind: "claimEvidence";
|
|
107
|
+
rule: Rule;
|
|
108
|
+
}
|
|
109
|
+
export type Classification = ClaimEvidenceClassification | DeterministicClassification | IfEditThenTestClassification | GitBranchPolicyClassification | CodeContentClassification | FileLifecycleClassification | NotARuleClassification | JudgmentClassification;
|
|
91
110
|
/**
|
|
92
111
|
* A rule is only treated as deterministic when it names a specific,
|
|
93
112
|
* literal, checkable token (a CLI flag, a command, an exact string) in
|
package/dist/checks/classify.js
CHANGED
|
@@ -60,7 +60,7 @@ const EVENT_RECORD_TITLE = /\b(incident|post-?mortem|retro(spective)?|outage|wha
|
|
|
60
60
|
* happens to name an incident; "Real incident (2026-08-28): ..." is a
|
|
61
61
|
* report that happens to contain the word never further along.
|
|
62
62
|
*/
|
|
63
|
-
const TITLE_OPENS_WITH_DIRECTIVE = /^\s*[-*+\d.\s]*(never|always|must|do not|don't|dont|avoid|ensure|prefer|only|make sure|be sure)\b/i;
|
|
63
|
+
const TITLE_OPENS_WITH_DIRECTIVE = /^\s*[-*+\d.\s]*(never|always|must|do not|don'?t|dont|avoid|ensure|prefer|only|make sure|be sure|no)\b/i;
|
|
64
64
|
function isEventRecord(rule) {
|
|
65
65
|
if (TITLE_OPENS_WITH_DIRECTIVE.test(rule.title))
|
|
66
66
|
return false;
|
|
@@ -134,6 +134,11 @@ function isCommandDocumentation(rule) {
|
|
|
134
134
|
return looksLikeBareCommand(rule.text);
|
|
135
135
|
}
|
|
136
136
|
function isNotARule(rule) {
|
|
137
|
+
// "No debug logging", "No force pushing" — one of the commonest ways a
|
|
138
|
+
// prohibition is written, and the directive list had no entry for it, so
|
|
139
|
+
// these were being dropped as documentation before any check saw them.
|
|
140
|
+
if (TITLE_OPENS_WITH_DIRECTIVE.test(rule.title))
|
|
141
|
+
return false;
|
|
137
142
|
// Checked before the directive test on purpose: an incident note that
|
|
138
143
|
// ends with its lesson contains a real directive, and would otherwise
|
|
139
144
|
// be enforced as though the history itself were the rule.
|
|
@@ -154,6 +159,55 @@ function isNotARule(rule) {
|
|
|
154
159
|
// as non-rules — caught by an existing test, not by inspection.
|
|
155
160
|
return !IMPERATIVE_INSTRUCTION.test(rule.title) && !IMPERATIVE_INSTRUCTION.test(rule.text);
|
|
156
161
|
}
|
|
162
|
+
/**
|
|
163
|
+
* A rule about not reporting something as done without the evidence.
|
|
164
|
+
*
|
|
165
|
+
* Requires BOTH halves in the same rule: a verb about reporting or
|
|
166
|
+
* claiming, and a noun about evidence or verification. Either alone is far
|
|
167
|
+
* too broad — "run the tests" has the second, "tell me what changed" has
|
|
168
|
+
* the first, and neither is this rule.
|
|
169
|
+
*
|
|
170
|
+
* These rules almost never carry a backtick literal, so without this they
|
|
171
|
+
* fall straight through to judgment. This project's own Rule 1 ("Evidence
|
|
172
|
+
* or it didn't happen") is exactly that shape, and it is mechanically
|
|
173
|
+
* answerable whenever the session both claimed and ran something.
|
|
174
|
+
*/
|
|
175
|
+
const REPORTING_VERB = /\b(?:report|claim|say|said|state|assert|tell|declar|announc|call(?:ing)?\s+it|mark(?:ing)?\s+it)\w*\b/i;
|
|
176
|
+
const EVIDENCE_NOUN = /\b(?:evidence|proof|prove|paste|pasted|verif\w*|receipt|output|actual\s+(?:result|output)|test\s+output)\b/i;
|
|
177
|
+
const DONE_WORD = /\b(?:done|complete\w*|pass(?:ing|ed|es)?|working|fixed|confirmed|success\w*|green|ready)\b/i;
|
|
178
|
+
/**
|
|
179
|
+
* A gate the agent must pass BEFORE acting, rather than a report it makes
|
|
180
|
+
* after. Real misroute found 2026-09-11: "Repeat-back before destructive or
|
|
181
|
+
* expensive actions" contains a reporting verb ("say"), an evidence noun
|
|
182
|
+
* ("paste evidence") and a done-word ("are done"), so it satisfied all
|
|
183
|
+
* three tests below and was then answered with a verdict about claiming a
|
|
184
|
+
* passing test suite — true of the session, and nothing to do with the rule.
|
|
185
|
+
*
|
|
186
|
+
* Routing wider than the checker's competence is worse than not routing at
|
|
187
|
+
* all: a confident, irrelevant answer costs more than an honest "this one
|
|
188
|
+
* is yours".
|
|
189
|
+
*/
|
|
190
|
+
const PRE_ACTION_GATE = /\b(?:repeat[- ]back|restate\s+what|wait\s+for\s+(?:confirmation|approval|explicit|sign[- ]?off)|ask\s+(?:first|before|for\s+permission)|get\s+(?:approval|sign[- ]?off|permission)|before\s+(?:you\s+)?(?:delet|overwrit|drop|truncat|wip|run|flip|deploy|push|commit|chang|modif|plac))\w*/i;
|
|
191
|
+
/**
|
|
192
|
+
* A body this long is a document section, not a rule.
|
|
193
|
+
*
|
|
194
|
+
* Rule bodies run to a median of 71 characters and a 99th percentile of
|
|
195
|
+
* about 1,700. Measured 2026-09-13, claimEvidence was routing sections with
|
|
196
|
+
* a median body of 1,932 and a longest of 16,502 — 51 of its 83 rules were
|
|
197
|
+
* over this threshold. A section that long contains a reporting verb, an
|
|
198
|
+
* evidence noun and a done-word somewhere by accident, which is how a rule
|
|
199
|
+
* titled "AutoEvolve Instructions for GitHub Copilot" produced 166 of 347
|
|
200
|
+
* failures on its own.
|
|
201
|
+
*/
|
|
202
|
+
const MAX_RULE_BODY_FOR_CLAIM = 1500;
|
|
203
|
+
function isClaimEvidenceRule(rule) {
|
|
204
|
+
if (rule.text.length > MAX_RULE_BODY_FOR_CLAIM)
|
|
205
|
+
return false;
|
|
206
|
+
const text = `${rule.title} ${rule.text}`;
|
|
207
|
+
if (PRE_ACTION_GATE.test(text))
|
|
208
|
+
return false;
|
|
209
|
+
return REPORTING_VERB.test(text) && EVIDENCE_NOUN.test(text) && DONE_WORD.test(text);
|
|
210
|
+
}
|
|
157
211
|
const BRANCH_WORD = /\bbranch\b/i;
|
|
158
212
|
// A function/method-call shape ("print(", "analytics.track(") is a strong,
|
|
159
213
|
// simple signal that a backtick literal names actual CODE, not a CLI
|
|
@@ -189,6 +243,35 @@ function isEditImpliesTestRule(rule) {
|
|
|
189
243
|
const text = `${rule.title} ${rule.text}`;
|
|
190
244
|
return EDIT_IMPLIES_TEST_PHRASE.test(text);
|
|
191
245
|
}
|
|
246
|
+
/**
|
|
247
|
+
* Function words that appear in backticks in real rules files but can never
|
|
248
|
+
* identify an action. Deliberately short and only English function words —
|
|
249
|
+
* `go`, `cd`, `rm`, `gh` are all real commands and must survive.
|
|
250
|
+
*/
|
|
251
|
+
const STOPWORD_LITERAL = new Set([
|
|
252
|
+
"a", "an", "and", "as", "at", "be", "by", "if", "in", "is", "it", "of",
|
|
253
|
+
"on", "or", "so", "the", "to", "we", "you", "this", "that", "with",
|
|
254
|
+
"for", "from", "not", "but", "are", "was", "do",
|
|
255
|
+
]);
|
|
256
|
+
/**
|
|
257
|
+
* Can this literal identify anything?
|
|
258
|
+
*
|
|
259
|
+
* Measured 2026-09-12 across 559 real rules files run against 5 real
|
|
260
|
+
* sessions — 2,795 reports, 18,025 verdicts, 967 of them FAIL. The corpus
|
|
261
|
+
* yields 267 literals that cannot: `,` appears 22 times and matches every
|
|
262
|
+
* file containing a comma; `/`, `:`, `!` match everything; `or`, `in`, `is`
|
|
263
|
+
* match ordinary prose. Each is a machine for confident accusations about
|
|
264
|
+
* nothing.
|
|
265
|
+
*
|
|
266
|
+
* Dropping the literal is not sufficient on its own — see classifyRule. If
|
|
267
|
+
* a rule has no usable literal left, the reason for calling it mechanical
|
|
268
|
+
* has gone with it, and it belongs in judgment.
|
|
269
|
+
*/
|
|
270
|
+
function isUsablePattern(token) {
|
|
271
|
+
if (!/[A-Za-z0-9]/.test(token))
|
|
272
|
+
return false;
|
|
273
|
+
return !STOPWORD_LITERAL.has(token.toLowerCase());
|
|
274
|
+
}
|
|
192
275
|
const BACKTICK_TOKEN = /`([^`]+)`/g;
|
|
193
276
|
// Keyword signal near the rule text that this is a mandatory action, not a
|
|
194
277
|
// ban. Deliberately conservative: only a short, explicit set of words a
|
|
@@ -196,8 +279,8 @@ const BACKTICK_TOKEN = /`([^`]+)`/g;
|
|
|
196
279
|
// ambiguous defaults to "forbid" semantics (pattern found = FAIL), which
|
|
197
280
|
// was the only behavior that existed before this — no regression risk,
|
|
198
281
|
// only an additive one for rules that clearly ask for a required action.
|
|
199
|
-
const REQUIRE_SIGNAL = /\b(always|must|required|require|ensure|need to)\b/i;
|
|
200
|
-
const FORBID_SIGNAL = /\b(never|don't|do not|forbidden|banned|must not)\b/
|
|
282
|
+
const REQUIRE_SIGNAL = /\b(always|must|required|require|ensure|need to|needs to|have to|has to|should|shall|make sure|be sure)\b/i;
|
|
283
|
+
const FORBID_SIGNAL = /\b(never|don'?t|do not|forbidden|banned|prohibited|disallow\w*|avoid|must not|should not|shouldn'?t|not allowed|no longer)\b|^\s*(?:#{1,6}\s*)?no\s+\S/im;
|
|
201
284
|
/**
|
|
202
285
|
* A rule can prohibit one thing AND prescribe another in the same breath:
|
|
203
286
|
* "NEVER squash when merging PRs. Use `gh pr merge --merge --admin`".
|
|
@@ -214,7 +297,7 @@ const FORBID_SIGNAL = /\b(never|don't|do not|forbidden|banned|must not)\b/i;
|
|
|
214
297
|
// often prescribes with a bare imperative ("Use `gh pr merge --merge`")
|
|
215
298
|
// rather than "you must use" — and it's exactly those imperative clauses
|
|
216
299
|
// whose literals get misattributed to the prohibiting half.
|
|
217
|
-
const PRESCRIPTIVE_VERB = /\b(use|run|prefer|apply|follow|call|invoke|stick to)\b/i;
|
|
300
|
+
const PRESCRIPTIVE_VERB = /\b(use|run|prefer|apply|follow|call|invoke|stick to|update|write|create|add|include|keep|maintain|document)\b/i;
|
|
218
301
|
function hasMixedPolarity(rule) {
|
|
219
302
|
const text = `${rule.title} ${rule.text}`;
|
|
220
303
|
if (!FORBID_SIGNAL.test(text))
|
|
@@ -236,6 +319,26 @@ function hasMixedPolarity(rule) {
|
|
|
236
319
|
const clauses = masked.split(/[.;\n]|(?:\s+-\s+)/).filter((c) => c.trim().length > 0);
|
|
237
320
|
return clauses.some((clause) => !FORBID_SIGNAL.test(clause) && (REQUIRE_SIGNAL.test(clause) || PRESCRIPTIVE_VERB.test(clause)));
|
|
238
321
|
}
|
|
322
|
+
/**
|
|
323
|
+
* The direction of a rule, or null when the text does not say.
|
|
324
|
+
*
|
|
325
|
+
* This used to default to "forbid" — the accusing direction — justified as
|
|
326
|
+
* carrying "no regression risk" because it was the pre-existing behaviour.
|
|
327
|
+
* It carried the worst risk there is. Measured 2026-09-12 against
|
|
328
|
+
* cypress-io/cypress's own CLAUDE.md:
|
|
329
|
+
*
|
|
330
|
+
* rule "After ANY correction from the user: update the closest
|
|
331
|
+
* relevant `CLAUDE.md` file"
|
|
332
|
+
* verdict FAIL — "CLAUDE.md" was actually modified
|
|
333
|
+
*
|
|
334
|
+
* The rule asks you to write that file. The session did. The tool failed it
|
|
335
|
+
* for complying, because no require-word appeared and the default took over.
|
|
336
|
+
*
|
|
337
|
+
* A polarity is what a person settles in one second and what a matcher must
|
|
338
|
+
* never assume: the two answers are opposites, and one of them accuses
|
|
339
|
+
* someone of doing what they were told to do. Undetermined now means the
|
|
340
|
+
* rule is not mechanically checkable and goes to judgment.
|
|
341
|
+
*/
|
|
239
342
|
function detectPolarity(rule) {
|
|
240
343
|
const text = `${rule.title} ${rule.text}`;
|
|
241
344
|
// An explicit forbid word anywhere wins over a require word — "you must
|
|
@@ -244,7 +347,30 @@ function detectPolarity(rule) {
|
|
|
244
347
|
return "forbid";
|
|
245
348
|
if (REQUIRE_SIGNAL.test(text))
|
|
246
349
|
return "require";
|
|
247
|
-
|
|
350
|
+
// A bare imperative — "Use `npm`", "Update the changelog" — is a
|
|
351
|
+
// requirement. Reading it as one is also the SAFE direction: a required
|
|
352
|
+
// pattern that never appears reports UNCLEAR, while a forbidden one that
|
|
353
|
+
// appears reports a violation. Guessing toward require can waste a check;
|
|
354
|
+
// guessing toward forbid accuses someone.
|
|
355
|
+
if (PRESCRIPTIVE_VERB.test(text))
|
|
356
|
+
return "require";
|
|
357
|
+
return null;
|
|
358
|
+
}
|
|
359
|
+
/**
|
|
360
|
+
* Was the direction READ from an explicit signal word, or INFERRED from a
|
|
361
|
+
* bare imperative?
|
|
362
|
+
*
|
|
363
|
+
* Named on the verdict rather than argued about. "Use npm for this project"
|
|
364
|
+
* is sometimes a required action and sometimes a description of current
|
|
365
|
+
* practice, and the only way to find out whether this inference is coverage
|
|
366
|
+
* or noise is to measure the two populations separately. Suggested on
|
|
367
|
+
* anthropics/claude-code#90542.
|
|
368
|
+
*/
|
|
369
|
+
function polarityWasInferred(rule) {
|
|
370
|
+
const text = `${rule.title} ${rule.text}`;
|
|
371
|
+
if (FORBID_SIGNAL.test(text) || REQUIRE_SIGNAL.test(text))
|
|
372
|
+
return false;
|
|
373
|
+
return PRESCRIPTIVE_VERB.test(text);
|
|
248
374
|
}
|
|
249
375
|
/**
|
|
250
376
|
* A rule is only treated as deterministic when it names a specific,
|
|
@@ -259,15 +385,22 @@ export function classifyRule(rule) {
|
|
|
259
385
|
if (isNotARule(rule)) {
|
|
260
386
|
return { kind: "notARule", rule };
|
|
261
387
|
}
|
|
388
|
+
// Checked before the backtick test below: these rules are about what the
|
|
389
|
+
// session SAID versus what it DID, and they essentially never name a
|
|
390
|
+
// literal, so they would otherwise fall through to judgment.
|
|
391
|
+
if (isClaimEvidenceRule(rule)) {
|
|
392
|
+
return { kind: "claimEvidence", rule };
|
|
393
|
+
}
|
|
394
|
+
// (length guard applied inside isClaimEvidenceRule)
|
|
262
395
|
const patterns = new Set();
|
|
263
396
|
for (const match of rule.text.matchAll(BACKTICK_TOKEN)) {
|
|
264
397
|
const token = match[1].trim();
|
|
265
|
-
if (token
|
|
398
|
+
if (isUsablePattern(token))
|
|
266
399
|
patterns.add(token);
|
|
267
400
|
}
|
|
268
401
|
for (const match of rule.title.matchAll(BACKTICK_TOKEN)) {
|
|
269
402
|
const token = match[1].trim();
|
|
270
|
-
if (token
|
|
403
|
+
if (isUsablePattern(token))
|
|
271
404
|
patterns.add(token);
|
|
272
405
|
}
|
|
273
406
|
// A rule that both forbids and prescribes can't be checked by literal
|
|
@@ -282,21 +415,34 @@ export function classifyRule(rule) {
|
|
|
282
415
|
return { kind: "judgment", rule };
|
|
283
416
|
}
|
|
284
417
|
const text = `${rule.title} ${rule.text}`;
|
|
418
|
+
// A rule whose direction cannot be read is not a rule this can check.
|
|
419
|
+
const polarity = detectPolarity(rule);
|
|
420
|
+
if (polarity === null) {
|
|
421
|
+
return { kind: "judgment", rule };
|
|
422
|
+
}
|
|
423
|
+
const polarityInferred = polarityWasInferred(rule);
|
|
285
424
|
if (BRANCH_WORD.test(text)) {
|
|
286
425
|
// first backtick literal is treated as the branch name — real rules
|
|
287
426
|
// this targets name exactly one branch ("the `demo` branch", "never
|
|
288
427
|
// push to `main`"), not a set of them
|
|
289
428
|
const [branchName] = patterns;
|
|
290
|
-
return { kind: "gitBranchPolicy", rule, branchName, polarity
|
|
429
|
+
return { kind: "gitBranchPolicy", rule, branchName, polarity, polarityInferred };
|
|
291
430
|
}
|
|
292
|
-
|
|
293
|
-
|
|
431
|
+
// Route on ANY code-shaped literal, but check ONLY the code-shaped ones.
|
|
432
|
+
// This used to pass every literal through once one of them looked like
|
|
433
|
+
// code, so a rule mentioning `foo(` and `name` searched written files for
|
|
434
|
+
// "name". Measured 2026-09-13: 456 of 2,871 code-content patterns were a
|
|
435
|
+
// single bare word — name, OK, FAIL, ERROR, Description — and each could
|
|
436
|
+
// fail any file containing it.
|
|
437
|
+
const codeLiterals = [...patterns].filter((p) => CODE_CONSTRUCT_PATTERN.test(p));
|
|
438
|
+
if (codeLiterals.length > 0) {
|
|
439
|
+
return { kind: "codeContent", rule, patterns: codeLiterals, polarity, polarityInferred };
|
|
294
440
|
}
|
|
295
441
|
const filePath = [...patterns].find((p) => FILE_PATH_PATTERN.test(p));
|
|
296
442
|
if (filePath && FILE_MUTATION_INTENT.test(text)) {
|
|
297
|
-
return { kind: "fileLifecycle", rule, filePath, polarity
|
|
443
|
+
return { kind: "fileLifecycle", rule, filePath, polarity, polarityInferred };
|
|
298
444
|
}
|
|
299
|
-
return { kind: "deterministic", rule, patterns: [...patterns], polarity
|
|
445
|
+
return { kind: "deterministic", rule, patterns: [...patterns], polarity, polarityInferred };
|
|
300
446
|
}
|
|
301
447
|
export function classifyRules(rules) {
|
|
302
448
|
return rules.map(classifyRule);
|