rulereceipt 0.1.19 → 0.1.21
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +39 -0
- package/dist/cli.js +64 -6
- package/dist/report/generateHtmlReport.js +50 -15
- package/dist/report/generateReport.js +50 -4
- package/dist/telemetry.d.ts +10 -0
- package/dist/telemetry.js +10 -0
- package/dist/types.d.ts +16 -0
- package/package.json +6 -6
package/README.md
CHANGED
|
@@ -1,8 +1,20 @@
|
|
|
1
1
|
# RuleReceipt
|
|
2
2
|
|
|
3
|
+
[](https://github.com/rulereceipt/rulereceipt/actions/workflows/ci.yml)
|
|
4
|
+
[](https://github.com/rulereceipt/rulereceipt/actions/workflows/codeql.yml)
|
|
5
|
+
[](https://scorecard.dev/viewer/?uri=github.com/rulereceipt/rulereceipt)
|
|
6
|
+
[](https://www.npmjs.com/package/rulereceipt)
|
|
7
|
+
[](https://www.npmjs.com/package/rulereceipt#provenance)
|
|
8
|
+
|
|
3
9
|
Checks whether a Claude Code session actually followed the rules in your
|
|
4
10
|
CLAUDE.md / AGENTS.md — with evidence, not just a vibe.
|
|
5
11
|
|
|
12
|
+
Every release from 0.1.19 on is built and published by GitHub Actions and
|
|
13
|
+
signed with [npm provenance](https://docs.npmjs.com/generating-provenance-statements),
|
|
14
|
+
so you can verify the published package was built from this repository at
|
|
15
|
+
a specific commit. No publishing token exists to be stolen. Check it
|
|
16
|
+
yourself with `npm audit signatures` after installing.
|
|
17
|
+
|
|
6
18
|
Licensed source-available software — see [LICENSE](./LICENSE) and
|
|
7
19
|
[NOTICE.md](./NOTICE.md) before reusing this code.
|
|
8
20
|
|
|
@@ -56,6 +68,31 @@ Published and live on npm, actively developed.
|
|
|
56
68
|
that exact file. (It proves the report matches the file, not that the
|
|
57
69
|
file is an unmodified record — see SECURITY.md.)
|
|
58
70
|
|
|
71
|
+
## Exit codes
|
|
72
|
+
|
|
73
|
+
`check` exits **1** when a rule was actually broken, and **0** otherwise,
|
|
74
|
+
so CI can gate on it. Rules that need human judgment report UNCLEAR and
|
|
75
|
+
never affect the exit code — most rules in a real CLAUDE.md need judgment,
|
|
76
|
+
and gating on those would make every build red on day one.
|
|
77
|
+
|
|
78
|
+
`--exit-zero` prints the report without failing the build. `--require-session`
|
|
79
|
+
does the opposite and is the one to use anywhere automated: it fails when
|
|
80
|
+
there is no session, or an empty one, instead of reporting a pass for a
|
|
81
|
+
check that never actually ran.
|
|
82
|
+
|
|
83
|
+
### A limit worth knowing before you wire this into CI
|
|
84
|
+
|
|
85
|
+
Claude Code writes its session transcript to the machine the agent ran on
|
|
86
|
+
— your laptop. A CI runner is a fresh machine that has never seen it, so a
|
|
87
|
+
CI job cannot check a session that happened on your laptop unless you
|
|
88
|
+
deliberately make that transcript available to the job. See
|
|
89
|
+
[templates/rulereceipt-ci.yml](./templates/rulereceipt-ci.yml), which
|
|
90
|
+
explains the options and, if you use it, fails loudly rather than passing
|
|
91
|
+
on a session it never found.
|
|
92
|
+
|
|
93
|
+
For most people the honest answer is simpler: run `rulereceipt check --html`
|
|
94
|
+
locally and attach the report to the PR.
|
|
95
|
+
|
|
59
96
|
## Sharing a report
|
|
60
97
|
|
|
61
98
|
`rulereceipt check --html` writes one self-contained HTML file. No
|
|
@@ -89,6 +126,8 @@ rulereceipt check # check the latest session in this project
|
|
|
89
126
|
rulereceipt check --markdown # same, formatted for pasting into a PR/Slack
|
|
90
127
|
rulereceipt check --html # write a shareable single-file HTML report you can send
|
|
91
128
|
rulereceipt check --html report.html # ...to a specific path
|
|
129
|
+
rulereceipt check --require-session # fail if there's no session, instead of passing silently
|
|
130
|
+
rulereceipt check --exit-zero # report failures without failing the build
|
|
92
131
|
rulereceipt check --llm # opt-in: grade judgment rules with your own Claude key
|
|
93
132
|
rulereceipt check --share # opt-in: send anonymous pass/fail/unclear counts
|
|
94
133
|
rulereceipt check --telemetry # opt-in: send one random per-machine ID
|
package/dist/cli.js
CHANGED
|
@@ -92,6 +92,7 @@ function needsLlmResult(rule) {
|
|
|
92
92
|
ruleTitle: rule.title,
|
|
93
93
|
ruleSource: rule.source,
|
|
94
94
|
status: "UNCLEAR",
|
|
95
|
+
needsHuman: true,
|
|
95
96
|
evidence: "NEEDS HUMAN REVIEW — this rule is a judgment call, not something that can be settled by looking at what commands ran. Read the session and decide for yourself. (`--llm` will give you a model's opinion on it, using your own Anthropic key — an opinion, not a verdict.)",
|
|
96
97
|
};
|
|
97
98
|
}
|
|
@@ -121,7 +122,7 @@ function writeHtmlReport(results, meta, cwd, target) {
|
|
|
121
122
|
}
|
|
122
123
|
}
|
|
123
124
|
async function runCheck(opts) {
|
|
124
|
-
const { markdown, share, email, emailAlways, llm, telemetry, html, transcriptOverride } = opts;
|
|
125
|
+
const { markdown, share, email, emailAlways, llm, telemetry, html, exitZero, requireSession, transcriptOverride } = opts;
|
|
125
126
|
const cwd = process.cwd();
|
|
126
127
|
const rules = loadRules(cwd);
|
|
127
128
|
if (rules.length === 0) {
|
|
@@ -139,9 +140,34 @@ async function runCheck(opts) {
|
|
|
139
140
|
console.log("No Claude Code session found for this project yet.\n" +
|
|
140
141
|
"Run Claude Code here at least once, then try `rulereceipt check` again — " +
|
|
141
142
|
"or pass --transcript <path-to-.jsonl> directly if your session lives somewhere non-standard.");
|
|
143
|
+
// Exiting 0 here is right for a person running this locally for the
|
|
144
|
+
// first time — nothing is wrong, there is simply nothing yet. It is
|
|
145
|
+
// dangerous anywhere automated, where a silent 0 reads as "checked,
|
|
146
|
+
// all clear" when nothing was checked at all. --require-session makes
|
|
147
|
+
// that case fail loudly. See the note in templates/rulereceipt-ci.yml.
|
|
148
|
+
if (requireSession) {
|
|
149
|
+
console.error("\n--require-session was set and no session was found, so nothing could be checked. " +
|
|
150
|
+
"Failing rather than reporting a pass for a check that never ran.");
|
|
151
|
+
process.exitCode = 1;
|
|
152
|
+
}
|
|
142
153
|
return;
|
|
143
154
|
}
|
|
144
155
|
const events = transcriptOverride ? readTranscriptFromFile(sessionFilePath) : readLatestTranscript(cwd);
|
|
156
|
+
// A session file with nothing in it produces a report full of PASSes,
|
|
157
|
+
// because no forbidden action appears in an empty session. That is
|
|
158
|
+
// technically true and badly misleading: "we found no proof of
|
|
159
|
+
// wrongdoing" gets printed as "you're fine." Same shape as the rule this
|
|
160
|
+
// tool already enforces on itself — an absence of evidence is not
|
|
161
|
+
// evidence. Say so out loud, and fail where a machine is reading it.
|
|
162
|
+
if (events.length === 0) {
|
|
163
|
+
console.log("\n⚠ This session file contains no recorded activity, so there was nothing to check against.\n" +
|
|
164
|
+
" Every result below reflects an empty session, not a clean one.");
|
|
165
|
+
if (requireSession) {
|
|
166
|
+
console.error("\n--require-session was set and the session was empty. Failing rather than reporting a pass for a check that had no evidence.");
|
|
167
|
+
process.exitCode = 1;
|
|
168
|
+
return;
|
|
169
|
+
}
|
|
170
|
+
}
|
|
145
171
|
const classifications = classifyRules(rules);
|
|
146
172
|
const deterministic = classifications.filter((c) => c.kind === "deterministic");
|
|
147
173
|
const ifEditThenTest = classifications.filter((c) => c.kind === "ifEditThenTest");
|
|
@@ -149,11 +175,22 @@ async function runCheck(opts) {
|
|
|
149
175
|
const codeContent = classifications.filter((c) => c.kind === "codeContent");
|
|
150
176
|
const fileLifecycle = classifications.filter((c) => c.kind === "fileLifecycle");
|
|
151
177
|
const judgment = classifications.filter((c) => c.kind === "judgment");
|
|
152
|
-
// Not rules at all — documentation, glossary entries, reference tables
|
|
153
|
-
//
|
|
154
|
-
//
|
|
155
|
-
//
|
|
156
|
-
//
|
|
178
|
+
// Not rules at all — documentation, glossary entries, reference tables,
|
|
179
|
+
// URLs, directory listings, code examples.
|
|
180
|
+
//
|
|
181
|
+
// Re-measured 2026-08-31 across 40 real public rule files: 943 of 1,441
|
|
182
|
+
// parsed items, i.e. 65.4%. A previous comment here said ~17%, which was
|
|
183
|
+
// wrong and made the tool look like it was discarding far less than it
|
|
184
|
+
// is. Sampled and reviewed by hand before trusting the new figure: the
|
|
185
|
+
// classification is correct, real rules files simply are mostly prose.
|
|
186
|
+
//
|
|
187
|
+
// The number that actually matters is the one about RULES rather than
|
|
188
|
+
// lines: of the 498 genuine rules in that corpus, 238 (47.8%) are
|
|
189
|
+
// mechanically answerable and 260 (52.2%) are judgment calls.
|
|
190
|
+
//
|
|
191
|
+
// Reported as a count so nothing is silently dropped, but never checked,
|
|
192
|
+
// since "did the session violate a directory listing" has no meaningful
|
|
193
|
+
// answer and any coincidental match is pure noise.
|
|
157
194
|
const notARule = classifications.filter((c) => c.kind === "notARule");
|
|
158
195
|
const deterministicResults = [
|
|
159
196
|
...runDeterministicChecks(deterministic, events),
|
|
@@ -201,6 +238,23 @@ async function runCheck(opts) {
|
|
|
201
238
|
if (isTelemetryEnabled(telemetry)) {
|
|
202
239
|
await sendTelemetryPing();
|
|
203
240
|
}
|
|
241
|
+
// Exit non-zero when a rule was actually broken, so CI can gate on it.
|
|
242
|
+
//
|
|
243
|
+
// This was a real shipped falsehood (found 2026-08-31):
|
|
244
|
+
// templates/rulereceipt-ci.yml told people to copy a workflow and said
|
|
245
|
+
// "rulereceipt already exits non-zero on FAIL, this just wires that
|
|
246
|
+
// into CI" — while `check` always exited 0. Anyone who used that
|
|
247
|
+
// template had a job that passed even as the agent broke their rules,
|
|
248
|
+
// which is worse than having no check at all, because it reads as
|
|
249
|
+
// evidence that nothing went wrong.
|
|
250
|
+
//
|
|
251
|
+
// Only FAIL counts. UNCLEAR must not, and that isn't a softening: most
|
|
252
|
+
// rules in a real CLAUDE.md need judgment, so without --llm they
|
|
253
|
+
// legitimately report UNCLEAR. Gating on those would make every build
|
|
254
|
+
// red on day one and the check would be deleted within a week.
|
|
255
|
+
if (!exitZero && results.some((r) => r.status === "FAIL")) {
|
|
256
|
+
process.exitCode = 1;
|
|
257
|
+
}
|
|
204
258
|
}
|
|
205
259
|
function runDemo(markdown) {
|
|
206
260
|
const meta = { sessionFilePath: null, ruleCount: DEMO_RESULTS.length };
|
|
@@ -222,6 +276,8 @@ program
|
|
|
222
276
|
.option("--llm", "opt-in: grade rules that need judgment (not just pattern matching) using your own Anthropic key. Without this flag, those rules report UNCLEAR and nothing is sent anywhere — deterministic checks always run with no key regardless.")
|
|
223
277
|
.option("--telemetry", "opt-in: send an anonymous install-count ping (a random per-machine ID, never rule text or results) so real distinct-install counts are knowable. Off by default. DO_NOT_TRACK=1 or RULERECEIPT_NO_TELEMETRY=1 overrides this flag back off.")
|
|
224
278
|
.option("--html [path]", `write a shareable single-file HTML report you can email, attach to a ticket, or print to PDF. Defaults to ./${DEFAULT_HTML_REPORT_NAME}. Written locally — nothing is uploaded.`)
|
|
279
|
+
.option("--exit-zero", "always exit 0, even when a rule was broken. Without this, `check` exits 1 on any FAIL so CI can gate on it (rules needing human judgment report UNCLEAR and never affect the exit code).")
|
|
280
|
+
.option("--require-session", "fail (exit 1) if no session is found, or the session is empty, instead of reporting a pass for a check that never actually ran. Use this anywhere automated.")
|
|
225
281
|
.option("--transcript <path>", "manual override: check this exact .jsonl session file instead of auto-detecting one. Useful if your Claude Code session lives somewhere non-standard that auto-detection doesn't cover.")
|
|
226
282
|
.action((opts) => {
|
|
227
283
|
runCheck({
|
|
@@ -233,6 +289,8 @@ program
|
|
|
233
289
|
telemetry: Boolean(opts.telemetry),
|
|
234
290
|
// commander gives `true` for a bare --html and the string for --html <path>
|
|
235
291
|
html: opts.html ?? false,
|
|
292
|
+
exitZero: Boolean(opts.exitZero),
|
|
293
|
+
requireSession: Boolean(opts.requireSession),
|
|
236
294
|
transcriptOverride: opts.transcript,
|
|
237
295
|
}).catch((err) => {
|
|
238
296
|
console.error("Something went wrong:", err instanceof Error ? err.message : err);
|
|
@@ -25,17 +25,32 @@ function stripControlChars(value) {
|
|
|
25
25
|
function clean(value) {
|
|
26
26
|
return escapeHtml(stripControlChars(value));
|
|
27
27
|
}
|
|
28
|
-
|
|
28
|
+
function bucketOf(result) {
|
|
29
|
+
if (result.status === "FAIL")
|
|
30
|
+
return "FAIL";
|
|
31
|
+
if (result.status === "PASS")
|
|
32
|
+
return "PASS";
|
|
33
|
+
return result.needsHuman ? "UNCLEAR_JUDGMENT" : "UNCLEAR_EVIDENCE";
|
|
34
|
+
}
|
|
35
|
+
const BUCKET_LABEL = {
|
|
29
36
|
FAIL: "Not followed",
|
|
37
|
+
UNCLEAR_EVIDENCE: "Couldn't tell",
|
|
38
|
+
UNCLEAR_JUDGMENT: "Needs your judgment",
|
|
30
39
|
PASS: "Followed",
|
|
31
|
-
|
|
40
|
+
};
|
|
41
|
+
const BUCKET_CLASS = {
|
|
42
|
+
FAIL: "fail",
|
|
43
|
+
UNCLEAR_EVIDENCE: "unclear",
|
|
44
|
+
UNCLEAR_JUDGMENT: "judgment",
|
|
45
|
+
PASS: "pass",
|
|
32
46
|
};
|
|
33
47
|
/**
|
|
34
48
|
* Failures first, deliberately. A report that opens with a wall of passes
|
|
35
49
|
* and buries one failure at the bottom is a report designed to be skimmed
|
|
36
|
-
* past — the opposite of what an audit document is for.
|
|
50
|
+
* past — the opposite of what an audit document is for. Judgment calls
|
|
51
|
+
* sit last: they're expected, and they're the longest section.
|
|
37
52
|
*/
|
|
38
|
-
const
|
|
53
|
+
const BUCKET_ORDER = ["FAIL", "UNCLEAR_EVIDENCE", "PASS", "UNCLEAR_JUDGMENT"];
|
|
39
54
|
function ruleLabel(result, all) {
|
|
40
55
|
const collides = all.filter((other) => other.ruleId === result.ruleId).length > 1;
|
|
41
56
|
return collides
|
|
@@ -46,24 +61,29 @@ function countBy(results, status) {
|
|
|
46
61
|
return results.filter((r) => r.status === status).length;
|
|
47
62
|
}
|
|
48
63
|
function renderResultRow(result, all) {
|
|
49
|
-
const
|
|
64
|
+
const bucket = bucketOf(result);
|
|
65
|
+
const cls = BUCKET_CLASS[bucket];
|
|
50
66
|
return `
|
|
51
67
|
<article class="result result--${cls}">
|
|
52
68
|
<div class="result__head">
|
|
53
|
-
<span class="badge badge--${cls}">${clean(
|
|
69
|
+
<span class="badge badge--${cls}">${clean(BUCKET_LABEL[bucket])}</span>
|
|
54
70
|
<span class="result__id">${clean(ruleLabel(result, all))}</span>
|
|
55
71
|
</div>
|
|
56
72
|
<h3 class="result__title">${clean(result.ruleTitle)}</h3>
|
|
57
73
|
${result.evidence ? `<p class="result__evidence">${clean(result.evidence)}</p>` : ""}
|
|
58
74
|
</article>`;
|
|
59
75
|
}
|
|
60
|
-
function renderSection(
|
|
61
|
-
const inSection = results.filter((r) => r
|
|
76
|
+
function renderSection(bucket, results, all) {
|
|
77
|
+
const inSection = results.filter((r) => bucketOf(r) === bucket);
|
|
62
78
|
if (inSection.length === 0)
|
|
63
79
|
return "";
|
|
80
|
+
const note = bucket === "UNCLEAR_JUDGMENT"
|
|
81
|
+
? `<p class="section__note">These were never questions a tool could settle — they need someone to read the session and decide. That is expected, not a gap in the check.</p>`
|
|
82
|
+
: "";
|
|
64
83
|
return `
|
|
65
84
|
<section class="section">
|
|
66
|
-
<h2 class="section__title">${clean(
|
|
85
|
+
<h2 class="section__title">${clean(BUCKET_LABEL[bucket])} <span class="section__count">${inSection.length}</span></h2>
|
|
86
|
+
${note}
|
|
67
87
|
${inSection.map((r) => renderResultRow(r, all)).join("")}
|
|
68
88
|
</section>`;
|
|
69
89
|
}
|
|
@@ -75,12 +95,22 @@ function renderSection(status, results, all) {
|
|
|
75
95
|
*/
|
|
76
96
|
function verdict(results) {
|
|
77
97
|
const fail = countBy(results, "FAIL");
|
|
78
|
-
const
|
|
98
|
+
const couldntTell = results.filter((r) => bucketOf(r) === "UNCLEAR_EVIDENCE").length;
|
|
99
|
+
const needsHuman = results.filter((r) => bucketOf(r) === "UNCLEAR_JUDGMENT").length;
|
|
79
100
|
if (fail > 0) {
|
|
80
101
|
return { text: `${fail} rule${fail === 1 ? "" : "s"} not followed`, cls: "fail" };
|
|
81
102
|
}
|
|
82
|
-
if (
|
|
83
|
-
return { text: `No rule violations found · ${
|
|
103
|
+
if (couldntTell > 0) {
|
|
104
|
+
return { text: `No rule violations found · ${couldntTell} couldn't be determined`, cls: "unclear" };
|
|
105
|
+
}
|
|
106
|
+
// Judgment rules still appear in the headline. They must not be styled
|
|
107
|
+
// as a problem — they aren't one — but they must not be omitted either.
|
|
108
|
+
// "No rule violations found" on its own, when half the rules were never
|
|
109
|
+
// evaluated mechanically, reads as full coverage. Naming the count is
|
|
110
|
+
// what keeps the headline from overclaiming, and it is the property the
|
|
111
|
+
// "does not claim a clean result" test exists to hold.
|
|
112
|
+
if (needsHuman > 0) {
|
|
113
|
+
return { text: `No rule violations found · ${needsHuman} still need human review`, cls: "judgment" };
|
|
84
114
|
}
|
|
85
115
|
return { text: "No rule violations found", cls: "pass" };
|
|
86
116
|
}
|
|
@@ -117,6 +147,7 @@ export function generateHtmlReport(results, meta) {
|
|
|
117
147
|
.verdict--fail { background: var(--fail-bg); border-color: #f0c4c6; }
|
|
118
148
|
.verdict--pass { background: var(--pass-bg); border-color: #bfe3cd; }
|
|
119
149
|
.verdict--unclear { background: var(--unclear-bg); border-color: #ecdcb0; }
|
|
150
|
+
.verdict--judgment { background: var(--panel); border-color: var(--line); }
|
|
120
151
|
.verdict strong { display: block; font-size: 19px; margin-bottom: 2px; }
|
|
121
152
|
.verdict span { color: var(--muted); font-size: 14px; }
|
|
122
153
|
.facts { width: 100%; border-collapse: collapse; margin-bottom: 32px; font-size: 14px; }
|
|
@@ -136,6 +167,9 @@ export function generateHtmlReport(results, meta) {
|
|
|
136
167
|
.badge--fail { background: var(--fail-bg); color: var(--fail); }
|
|
137
168
|
.badge--pass { background: var(--pass-bg); color: var(--pass); }
|
|
138
169
|
.badge--unclear { background: var(--unclear-bg); color: var(--unclear); }
|
|
170
|
+
.result--judgment { border-left-color: #6b6f76; }
|
|
171
|
+
.badge--judgment { background: #f2f3f5; color: #4a4e55; }
|
|
172
|
+
.section__note { font-size: 13px; color: var(--muted); margin: -4px 0 12px; }
|
|
139
173
|
.result__id { font-size: 12px; color: var(--muted); }
|
|
140
174
|
.result__title { font-size: 15px; margin: 0 0 6px; font-weight: 600; }
|
|
141
175
|
.result__evidence { margin: 0; font-size: 14px; color: var(--muted); white-space: pre-wrap; }
|
|
@@ -166,7 +200,7 @@ export function generateHtmlReport(results, meta) {
|
|
|
166
200
|
|
|
167
201
|
<div class="verdict verdict--${v.cls}">
|
|
168
202
|
<strong>${clean(v.text)}</strong>
|
|
169
|
-
<span>${countBy(results, "PASS")} followed · ${countBy(results, "FAIL")} not followed · ${
|
|
203
|
+
<span>${countBy(results, "PASS")} followed · ${countBy(results, "FAIL")} not followed · ${results.filter((r) => bucketOf(r) === "UNCLEAR_EVIDENCE").length} couldn't tell · ${results.filter((r) => bucketOf(r) === "UNCLEAR_JUDGMENT").length} need your judgment</span>
|
|
170
204
|
</div>
|
|
171
205
|
|
|
172
206
|
<table class="facts">
|
|
@@ -177,14 +211,15 @@ export function generateHtmlReport(results, meta) {
|
|
|
177
211
|
<tr><th>Tool version</th><td><code>rulereceipt ${clean(meta.toolVersion)}</code></td></tr>
|
|
178
212
|
</table>
|
|
179
213
|
|
|
180
|
-
${
|
|
214
|
+
${BUCKET_ORDER.map((b) => renderSection(b, results, results)).join("")}
|
|
181
215
|
|
|
182
216
|
<div class="note">
|
|
183
217
|
<h2>How to read this report</h2>
|
|
184
218
|
<ul>
|
|
185
219
|
<li><strong>Followed</strong> — a specific action in the session satisfies the rule, or the forbidden action never occurred.</li>
|
|
186
220
|
<li><strong>Not followed</strong> — a real action in the session contradicts the rule. The evidence quotes it.</li>
|
|
187
|
-
<li><strong>
|
|
221
|
+
<li><strong>Couldn't tell</strong> — the check ran and the evidence was ambiguous. A genuine gap.</li>
|
|
222
|
+
<li><strong>Needs your judgment</strong> — this rule never had a mechanical answer ("surface bad news first"). Not a pass, not a failure, and not a shortcoming of the check: it is the part that was always a person's call.</li>
|
|
188
223
|
</ul>
|
|
189
224
|
<h2>What this report does not establish</h2>
|
|
190
225
|
<ul>
|
|
@@ -44,11 +44,32 @@ function ruleLabel(r, results) {
|
|
|
44
44
|
const collides = results.filter((other) => other.ruleId === r.ruleId).length > 1;
|
|
45
45
|
return collides ? `Rule ${r.ruleId} (${r.ruleSource}) — ${r.ruleTitle}` : `Rule ${r.ruleId} — ${r.ruleTitle}`;
|
|
46
46
|
}
|
|
47
|
+
/**
|
|
48
|
+
* Separates the two very different things that were both being reported
|
|
49
|
+
* as "unclear":
|
|
50
|
+
*
|
|
51
|
+
* - couldn't tell — the tool looked and the evidence was ambiguous.
|
|
52
|
+
* - needs you — a judgment call that never had a mechanical
|
|
53
|
+
* answer ("surface bad news first").
|
|
54
|
+
*
|
|
55
|
+
* Collapsing them made the tool look like it failed on most rules. It
|
|
56
|
+
* doesn't: across 40 real rules files, 52.2% of the actual rules people
|
|
57
|
+
* write are judgment calls. Those aren't the tool falling short, they're
|
|
58
|
+
* the half of the work that was always a human's. Naming that honestly is
|
|
59
|
+
* the difference between a report that reads as broken and one that reads
|
|
60
|
+
* as a division of labour.
|
|
61
|
+
*/
|
|
47
62
|
function summaryLine(results) {
|
|
48
63
|
const pass = results.filter((r) => r.status === "PASS").length;
|
|
49
64
|
const fail = results.filter((r) => r.status === "FAIL").length;
|
|
50
|
-
const
|
|
51
|
-
|
|
65
|
+
const needsHuman = results.filter((r) => r.status === "UNCLEAR" && r.needsHuman).length;
|
|
66
|
+
const couldntTell = results.filter((r) => r.status === "UNCLEAR" && !r.needsHuman).length;
|
|
67
|
+
const parts = [`${pass} followed`, `${fail} not followed`];
|
|
68
|
+
if (couldntTell > 0)
|
|
69
|
+
parts.push(`${couldntTell} couldn't tell`);
|
|
70
|
+
if (needsHuman > 0)
|
|
71
|
+
parts.push(`${needsHuman} need your judgment`);
|
|
72
|
+
return parts.join(" · ");
|
|
52
73
|
}
|
|
53
74
|
export function generateReport(results, meta) {
|
|
54
75
|
const clean = results.map(sanitize);
|
|
@@ -67,6 +88,31 @@ export function generateReport(results, meta) {
|
|
|
67
88
|
lines.push(`checked: ${new Date().toISOString()}`);
|
|
68
89
|
return lines.join("\n");
|
|
69
90
|
}
|
|
91
|
+
/**
|
|
92
|
+
* Makes a value safe to place inside a markdown table cell.
|
|
93
|
+
*
|
|
94
|
+
* The previous version escaped only `|`, which broke a real table three
|
|
95
|
+
* separate ways (first found by CodeQL js/incomplete-sanitization, the
|
|
96
|
+
* other two while fixing it):
|
|
97
|
+
*
|
|
98
|
+
* 1. Backslash was not escaped, so evidence containing a literal `\|`
|
|
99
|
+
* became `\\|` — rendering as a backslash followed by a live column
|
|
100
|
+
* separator, splitting the row. Backslash must be escaped FIRST, or
|
|
101
|
+
* it re-escapes the pipes added afterwards.
|
|
102
|
+
* 2. The rule TITLE was not escaped at all, so any rule whose title
|
|
103
|
+
* contains a pipe broke the table. Titles come from a user's
|
|
104
|
+
* CLAUDE.md, and pipes appear naturally in shell examples.
|
|
105
|
+
* 3. Newlines were not handled. stripControlChars deliberately keeps
|
|
106
|
+
* `\n` so multi-line evidence stays readable in the terminal, but a
|
|
107
|
+
* newline inside a table cell ends the row and destroys everything
|
|
108
|
+
* below it. Rendered as a literal <br> instead.
|
|
109
|
+
*/
|
|
110
|
+
function escapeMarkdownCell(value) {
|
|
111
|
+
return value
|
|
112
|
+
.replace(/\\/g, "\\\\")
|
|
113
|
+
.replace(/\|/g, "\\|")
|
|
114
|
+
.replace(/\r?\n/g, "<br>");
|
|
115
|
+
}
|
|
70
116
|
export function generateMarkdownReport(results, meta) {
|
|
71
117
|
const clean = results.map(sanitize);
|
|
72
118
|
const lines = [];
|
|
@@ -75,8 +121,8 @@ export function generateMarkdownReport(results, meta) {
|
|
|
75
121
|
lines.push("| Status | Rule | Evidence |");
|
|
76
122
|
lines.push("|---|---|---|");
|
|
77
123
|
for (const r of clean) {
|
|
78
|
-
const evidence = (r.evidence || "")
|
|
79
|
-
lines.push(`| ${MARK[r.status]} ${r.status} | ${ruleLabel(r, clean)} | ${evidence} |`);
|
|
124
|
+
const evidence = escapeMarkdownCell(r.evidence || "");
|
|
125
|
+
lines.push(`| ${MARK[r.status]} ${r.status} | ${escapeMarkdownCell(ruleLabel(r, clean))} | ${evidence} |`);
|
|
80
126
|
}
|
|
81
127
|
lines.push("");
|
|
82
128
|
const hash = computeTranscriptHash(meta.sessionFilePath);
|
package/dist/telemetry.d.ts
CHANGED
|
@@ -15,5 +15,15 @@ export declare function isTelemetryEnabled(telemetryFlag: boolean): boolean;
|
|
|
15
15
|
* session content, or even pass/fail counts (that's what opt-in --share is
|
|
16
16
|
* for). A send failure must never affect the `check` command's own exit
|
|
17
17
|
* code or output; this is best-effort and silent on failure.
|
|
18
|
+
*
|
|
19
|
+
* CodeQL flags this as js/file-access-to-http ("outbound network request
|
|
20
|
+
* depends on file data"), which is technically accurate and not a real
|
|
21
|
+
* issue: the file it reads is ~/.rulereceipt/telemetry-id, whose entire
|
|
22
|
+
* contents are a random UUID this tool generated and wrote itself. No
|
|
23
|
+
* user content, no path, and nothing derived from the session ever
|
|
24
|
+
* reaches this request. Reviewed and dismissed deliberately rather than
|
|
25
|
+
* left open — a permanently red alert list is one nobody reads. If the
|
|
26
|
+
* payload here ever grows beyond `{ id }`, that decision is void and
|
|
27
|
+
* this needs re-reviewing.
|
|
18
28
|
*/
|
|
19
29
|
export declare function sendTelemetryPing(): Promise<void>;
|
package/dist/telemetry.js
CHANGED
|
@@ -61,6 +61,16 @@ export function isTelemetryEnabled(telemetryFlag) {
|
|
|
61
61
|
* session content, or even pass/fail counts (that's what opt-in --share is
|
|
62
62
|
* for). A send failure must never affect the `check` command's own exit
|
|
63
63
|
* code or output; this is best-effort and silent on failure.
|
|
64
|
+
*
|
|
65
|
+
* CodeQL flags this as js/file-access-to-http ("outbound network request
|
|
66
|
+
* depends on file data"), which is technically accurate and not a real
|
|
67
|
+
* issue: the file it reads is ~/.rulereceipt/telemetry-id, whose entire
|
|
68
|
+
* contents are a random UUID this tool generated and wrote itself. No
|
|
69
|
+
* user content, no path, and nothing derived from the session ever
|
|
70
|
+
* reaches this request. Reviewed and dismissed deliberately rather than
|
|
71
|
+
* left open — a permanently red alert list is one nobody reads. If the
|
|
72
|
+
* payload here ever grows beyond `{ id }`, that decision is void and
|
|
73
|
+
* this needs re-reviewing.
|
|
64
74
|
*/
|
|
65
75
|
export async function sendTelemetryPing() {
|
|
66
76
|
const id = getOrCreateTelemetryId();
|
package/dist/types.d.ts
CHANGED
|
@@ -32,4 +32,20 @@ export interface CheckResult {
|
|
|
32
32
|
ruleSource: "global" | "project";
|
|
33
33
|
status: CheckStatus;
|
|
34
34
|
evidence: string;
|
|
35
|
+
/**
|
|
36
|
+
* True when this rule was never mechanically answerable — a judgment
|
|
37
|
+
* call like "surface bad news first", which has no command to inspect.
|
|
38
|
+
*
|
|
39
|
+
* Exists because collapsing these into a single UNCLEAR count made the
|
|
40
|
+
* tool look broken. Measured across 40 real rules files: of the actual
|
|
41
|
+
* rules people write, 47.8% are mechanically answerable and 52.2% are
|
|
42
|
+
* judgment calls. Reporting "1 pass · 0 fail · 14 unclear" reads as
|
|
43
|
+
* fourteen failures, when most of those were a human's call from the
|
|
44
|
+
* start and the tool is working exactly as intended.
|
|
45
|
+
*
|
|
46
|
+
* "I could not determine this" and "this was always yours to decide"
|
|
47
|
+
* are different statements, and a tool about honest reporting should
|
|
48
|
+
* not blur them.
|
|
49
|
+
*/
|
|
50
|
+
needsHuman?: boolean;
|
|
35
51
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "rulereceipt",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.21",
|
|
4
4
|
"description": "Checks whether a Claude Code session actually followed your CLAUDE.md / AGENTS.md rules, with evidence.",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
@@ -45,16 +45,16 @@
|
|
|
45
45
|
"license": "SEE LICENSE IN LICENSE",
|
|
46
46
|
"dependencies": {
|
|
47
47
|
"@anthropic-ai/sdk": "^0.32.0",
|
|
48
|
-
"commander": "^
|
|
49
|
-
"nodemailer": "^9.0.
|
|
48
|
+
"commander": "^15.0.0",
|
|
49
|
+
"nodemailer": "^9.0.6"
|
|
50
50
|
},
|
|
51
51
|
"devDependencies": {
|
|
52
|
-
"@types/node": "^
|
|
52
|
+
"@types/node": "^26.4.0",
|
|
53
53
|
"@types/nodemailer": "^8.0.1",
|
|
54
|
-
"eslint": "^9.
|
|
54
|
+
"eslint": "^10.9.1",
|
|
55
55
|
"tsx": "^4.19.0",
|
|
56
56
|
"typescript": "^5.6.0",
|
|
57
|
-
"typescript-eslint": "^8.
|
|
57
|
+
"typescript-eslint": "^8.68.0",
|
|
58
58
|
"vitest": "^4.1.11"
|
|
59
59
|
}
|
|
60
60
|
}
|