rulereceipt 0.1.20 → 0.1.22
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +25 -5
- package/dist/report/generateHtmlReport.js +76 -16
- package/dist/report/generateReport.js +23 -2
- package/dist/types.d.ts +16 -0
- package/package.json +1 -1
package/dist/cli.js
CHANGED
|
@@ -92,6 +92,7 @@ function needsLlmResult(rule) {
|
|
|
92
92
|
ruleTitle: rule.title,
|
|
93
93
|
ruleSource: rule.source,
|
|
94
94
|
status: "UNCLEAR",
|
|
95
|
+
needsHuman: true,
|
|
95
96
|
evidence: "NEEDS HUMAN REVIEW — this rule is a judgment call, not something that can be settled by looking at what commands ran. Read the session and decide for yourself. (`--llm` will give you a model's opinion on it, using your own Anthropic key — an opinion, not a verdict.)",
|
|
96
97
|
};
|
|
97
98
|
}
|
|
@@ -115,6 +116,14 @@ function writeHtmlReport(results, meta, cwd, target) {
|
|
|
115
116
|
writeFileSync(outPath, html, "utf-8");
|
|
116
117
|
console.log(`\nShareable report written to ${outPath}`);
|
|
117
118
|
console.log("Open it in a browser, attach it to an email, or print it to PDF. It's a single self-contained file.");
|
|
119
|
+
// Found by dogfooding on a real session (2026-08-31): this file
|
|
120
|
+
// reproduces rule text and quoted evidence verbatim, which is exactly
|
|
121
|
+
// what makes it useful — and means it inherits whatever is in the
|
|
122
|
+
// rules file. A real CLAUDE.md turned out to contain an employer
|
|
123
|
+
// name, an office email, and absolute paths. Home paths are redacted
|
|
124
|
+
// automatically; nothing else can be, so say so plainly at the moment
|
|
125
|
+
// the file is created rather than burying it in a policy page.
|
|
126
|
+
console.log("Read it before you send it: it quotes your rule text and session evidence verbatim, so anything sensitive in your CLAUDE.md is in there too. (Home paths are shortened to ~.)");
|
|
118
127
|
}
|
|
119
128
|
catch (err) {
|
|
120
129
|
console.log(`\n(--html: couldn't write ${outPath} — ${err instanceof Error ? err.message : String(err)})`);
|
|
@@ -174,11 +183,22 @@ async function runCheck(opts) {
|
|
|
174
183
|
const codeContent = classifications.filter((c) => c.kind === "codeContent");
|
|
175
184
|
const fileLifecycle = classifications.filter((c) => c.kind === "fileLifecycle");
|
|
176
185
|
const judgment = classifications.filter((c) => c.kind === "judgment");
|
|
177
|
-
// Not rules at all — documentation, glossary entries, reference tables
|
|
178
|
-
//
|
|
179
|
-
//
|
|
180
|
-
//
|
|
181
|
-
//
|
|
186
|
+
// Not rules at all — documentation, glossary entries, reference tables,
|
|
187
|
+
// URLs, directory listings, code examples.
|
|
188
|
+
//
|
|
189
|
+
// Re-measured 2026-08-31 across 40 real public rule files: 943 of 1,441
|
|
190
|
+
// parsed items, i.e. 65.4%. A previous comment here said ~17%, which was
|
|
191
|
+
// wrong and made the tool look like it was discarding far less than it
|
|
192
|
+
// is. Sampled and reviewed by hand before trusting the new figure: the
|
|
193
|
+
// classification is correct, real rules files simply are mostly prose.
|
|
194
|
+
//
|
|
195
|
+
// The number that actually matters is the one about RULES rather than
|
|
196
|
+
// lines: of the 498 genuine rules in that corpus, 238 (47.8%) are
|
|
197
|
+
// mechanically answerable and 260 (52.2%) are judgment calls.
|
|
198
|
+
//
|
|
199
|
+
// Reported as a count so nothing is silently dropped, but never checked,
|
|
200
|
+
// since "did the session violate a directory listing" has no meaningful
|
|
201
|
+
// answer and any coincidental match is pure noise.
|
|
182
202
|
const notARule = classifications.filter((c) => c.kind === "notARule");
|
|
183
203
|
const deterministicResults = [
|
|
184
204
|
...runDeterministicChecks(deterministic, events),
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { basename } from "node:path";
|
|
2
|
+
import { homedir } from "node:os";
|
|
2
3
|
import { computeTranscriptHash } from "./generateReport.js";
|
|
3
4
|
/**
|
|
4
5
|
* Escapes the five characters that can break out of either an HTML text
|
|
@@ -21,21 +22,60 @@ function escapeHtml(value) {
|
|
|
21
22
|
function stripControlChars(value) {
|
|
22
23
|
return value.replace(/[\x00-\x09\x0B-\x1F\x7F-\x9F]/g, "");
|
|
23
24
|
}
|
|
25
|
+
/**
|
|
26
|
+
* Replaces the user's home directory with `~`.
|
|
27
|
+
*
|
|
28
|
+
* Found by dogfooding on a real session (2026-08-31): the shareable
|
|
29
|
+
* report is the one output explicitly designed to be emailed to someone
|
|
30
|
+
* else, and it was printing absolute paths like
|
|
31
|
+
* /Users/<realname>/Desktop/... in both the project header and inside
|
|
32
|
+
* quoted evidence. For anyone publishing under a pseudonym — or simply
|
|
33
|
+
* anyone who would rather not hand a stranger their username and
|
|
34
|
+
* directory layout — that is a leak in the one artifact most likely to
|
|
35
|
+
* leave the machine.
|
|
36
|
+
*
|
|
37
|
+
* This is a mechanical, judgment-free redaction: it removes the home
|
|
38
|
+
* prefix and nothing else. It is NOT a general secret scrubber, and must
|
|
39
|
+
* not be described as one. Rule text and evidence are still reproduced
|
|
40
|
+
* verbatim, because that is what makes the report useful — which is why
|
|
41
|
+
* the CLI warns the user to read the file before sending it.
|
|
42
|
+
*/
|
|
43
|
+
function redactHome(value) {
|
|
44
|
+
const home = homedir();
|
|
45
|
+
if (!home || home === "/" || home.length < 4)
|
|
46
|
+
return value;
|
|
47
|
+
return value.split(home).join("~");
|
|
48
|
+
}
|
|
24
49
|
/** Single choke point: nothing untrusted reaches the document except through this. */
|
|
25
50
|
function clean(value) {
|
|
26
|
-
return escapeHtml(stripControlChars(value));
|
|
51
|
+
return escapeHtml(stripControlChars(redactHome(value)));
|
|
27
52
|
}
|
|
28
|
-
|
|
53
|
+
function bucketOf(result) {
|
|
54
|
+
if (result.status === "FAIL")
|
|
55
|
+
return "FAIL";
|
|
56
|
+
if (result.status === "PASS")
|
|
57
|
+
return "PASS";
|
|
58
|
+
return result.needsHuman ? "UNCLEAR_JUDGMENT" : "UNCLEAR_EVIDENCE";
|
|
59
|
+
}
|
|
60
|
+
const BUCKET_LABEL = {
|
|
29
61
|
FAIL: "Not followed",
|
|
62
|
+
UNCLEAR_EVIDENCE: "Couldn't tell",
|
|
63
|
+
UNCLEAR_JUDGMENT: "Needs your judgment",
|
|
30
64
|
PASS: "Followed",
|
|
31
|
-
|
|
65
|
+
};
|
|
66
|
+
const BUCKET_CLASS = {
|
|
67
|
+
FAIL: "fail",
|
|
68
|
+
UNCLEAR_EVIDENCE: "unclear",
|
|
69
|
+
UNCLEAR_JUDGMENT: "judgment",
|
|
70
|
+
PASS: "pass",
|
|
32
71
|
};
|
|
33
72
|
/**
|
|
34
73
|
* Failures first, deliberately. A report that opens with a wall of passes
|
|
35
74
|
* and buries one failure at the bottom is a report designed to be skimmed
|
|
36
|
-
* past — the opposite of what an audit document is for.
|
|
75
|
+
* past — the opposite of what an audit document is for. Judgment calls
|
|
76
|
+
* sit last: they're expected, and they're the longest section.
|
|
37
77
|
*/
|
|
38
|
-
const
|
|
78
|
+
const BUCKET_ORDER = ["FAIL", "UNCLEAR_EVIDENCE", "PASS", "UNCLEAR_JUDGMENT"];
|
|
39
79
|
function ruleLabel(result, all) {
|
|
40
80
|
const collides = all.filter((other) => other.ruleId === result.ruleId).length > 1;
|
|
41
81
|
return collides
|
|
@@ -46,24 +86,29 @@ function countBy(results, status) {
|
|
|
46
86
|
return results.filter((r) => r.status === status).length;
|
|
47
87
|
}
|
|
48
88
|
function renderResultRow(result, all) {
|
|
49
|
-
const
|
|
89
|
+
const bucket = bucketOf(result);
|
|
90
|
+
const cls = BUCKET_CLASS[bucket];
|
|
50
91
|
return `
|
|
51
92
|
<article class="result result--${cls}">
|
|
52
93
|
<div class="result__head">
|
|
53
|
-
<span class="badge badge--${cls}">${clean(
|
|
94
|
+
<span class="badge badge--${cls}">${clean(BUCKET_LABEL[bucket])}</span>
|
|
54
95
|
<span class="result__id">${clean(ruleLabel(result, all))}</span>
|
|
55
96
|
</div>
|
|
56
97
|
<h3 class="result__title">${clean(result.ruleTitle)}</h3>
|
|
57
98
|
${result.evidence ? `<p class="result__evidence">${clean(result.evidence)}</p>` : ""}
|
|
58
99
|
</article>`;
|
|
59
100
|
}
|
|
60
|
-
function renderSection(
|
|
61
|
-
const inSection = results.filter((r) => r
|
|
101
|
+
function renderSection(bucket, results, all) {
|
|
102
|
+
const inSection = results.filter((r) => bucketOf(r) === bucket);
|
|
62
103
|
if (inSection.length === 0)
|
|
63
104
|
return "";
|
|
105
|
+
const note = bucket === "UNCLEAR_JUDGMENT"
|
|
106
|
+
? `<p class="section__note">These were never questions a tool could settle — they need someone to read the session and decide. That is expected, not a gap in the check.</p>`
|
|
107
|
+
: "";
|
|
64
108
|
return `
|
|
65
109
|
<section class="section">
|
|
66
|
-
<h2 class="section__title">${clean(
|
|
110
|
+
<h2 class="section__title">${clean(BUCKET_LABEL[bucket])} <span class="section__count">${inSection.length}</span></h2>
|
|
111
|
+
${note}
|
|
67
112
|
${inSection.map((r) => renderResultRow(r, all)).join("")}
|
|
68
113
|
</section>`;
|
|
69
114
|
}
|
|
@@ -75,12 +120,22 @@ function renderSection(status, results, all) {
|
|
|
75
120
|
*/
|
|
76
121
|
function verdict(results) {
|
|
77
122
|
const fail = countBy(results, "FAIL");
|
|
78
|
-
const
|
|
123
|
+
const couldntTell = results.filter((r) => bucketOf(r) === "UNCLEAR_EVIDENCE").length;
|
|
124
|
+
const needsHuman = results.filter((r) => bucketOf(r) === "UNCLEAR_JUDGMENT").length;
|
|
79
125
|
if (fail > 0) {
|
|
80
126
|
return { text: `${fail} rule${fail === 1 ? "" : "s"} not followed`, cls: "fail" };
|
|
81
127
|
}
|
|
82
|
-
if (
|
|
83
|
-
return { text: `No rule violations found · ${
|
|
128
|
+
if (couldntTell > 0) {
|
|
129
|
+
return { text: `No rule violations found · ${couldntTell} couldn't be determined`, cls: "unclear" };
|
|
130
|
+
}
|
|
131
|
+
// Judgment rules still appear in the headline. They must not be styled
|
|
132
|
+
// as a problem — they aren't one — but they must not be omitted either.
|
|
133
|
+
// "No rule violations found" on its own, when half the rules were never
|
|
134
|
+
// evaluated mechanically, reads as full coverage. Naming the count is
|
|
135
|
+
// what keeps the headline from overclaiming, and it is the property the
|
|
136
|
+
// "does not claim a clean result" test exists to hold.
|
|
137
|
+
if (needsHuman > 0) {
|
|
138
|
+
return { text: `No rule violations found · ${needsHuman} still need human review`, cls: "judgment" };
|
|
84
139
|
}
|
|
85
140
|
return { text: "No rule violations found", cls: "pass" };
|
|
86
141
|
}
|
|
@@ -117,6 +172,7 @@ export function generateHtmlReport(results, meta) {
|
|
|
117
172
|
.verdict--fail { background: var(--fail-bg); border-color: #f0c4c6; }
|
|
118
173
|
.verdict--pass { background: var(--pass-bg); border-color: #bfe3cd; }
|
|
119
174
|
.verdict--unclear { background: var(--unclear-bg); border-color: #ecdcb0; }
|
|
175
|
+
.verdict--judgment { background: var(--panel); border-color: var(--line); }
|
|
120
176
|
.verdict strong { display: block; font-size: 19px; margin-bottom: 2px; }
|
|
121
177
|
.verdict span { color: var(--muted); font-size: 14px; }
|
|
122
178
|
.facts { width: 100%; border-collapse: collapse; margin-bottom: 32px; font-size: 14px; }
|
|
@@ -136,6 +192,9 @@ export function generateHtmlReport(results, meta) {
|
|
|
136
192
|
.badge--fail { background: var(--fail-bg); color: var(--fail); }
|
|
137
193
|
.badge--pass { background: var(--pass-bg); color: var(--pass); }
|
|
138
194
|
.badge--unclear { background: var(--unclear-bg); color: var(--unclear); }
|
|
195
|
+
.result--judgment { border-left-color: #6b6f76; }
|
|
196
|
+
.badge--judgment { background: #f2f3f5; color: #4a4e55; }
|
|
197
|
+
.section__note { font-size: 13px; color: var(--muted); margin: -4px 0 12px; }
|
|
139
198
|
.result__id { font-size: 12px; color: var(--muted); }
|
|
140
199
|
.result__title { font-size: 15px; margin: 0 0 6px; font-weight: 600; }
|
|
141
200
|
.result__evidence { margin: 0; font-size: 14px; color: var(--muted); white-space: pre-wrap; }
|
|
@@ -166,7 +225,7 @@ export function generateHtmlReport(results, meta) {
|
|
|
166
225
|
|
|
167
226
|
<div class="verdict verdict--${v.cls}">
|
|
168
227
|
<strong>${clean(v.text)}</strong>
|
|
169
|
-
<span>${countBy(results, "PASS")} followed · ${countBy(results, "FAIL")} not followed · ${
|
|
228
|
+
<span>${countBy(results, "PASS")} followed · ${countBy(results, "FAIL")} not followed · ${results.filter((r) => bucketOf(r) === "UNCLEAR_EVIDENCE").length} couldn't tell · ${results.filter((r) => bucketOf(r) === "UNCLEAR_JUDGMENT").length} need your judgment</span>
|
|
170
229
|
</div>
|
|
171
230
|
|
|
172
231
|
<table class="facts">
|
|
@@ -177,14 +236,15 @@ export function generateHtmlReport(results, meta) {
|
|
|
177
236
|
<tr><th>Tool version</th><td><code>rulereceipt ${clean(meta.toolVersion)}</code></td></tr>
|
|
178
237
|
</table>
|
|
179
238
|
|
|
180
|
-
${
|
|
239
|
+
${BUCKET_ORDER.map((b) => renderSection(b, results, results)).join("")}
|
|
181
240
|
|
|
182
241
|
<div class="note">
|
|
183
242
|
<h2>How to read this report</h2>
|
|
184
243
|
<ul>
|
|
185
244
|
<li><strong>Followed</strong> — a specific action in the session satisfies the rule, or the forbidden action never occurred.</li>
|
|
186
245
|
<li><strong>Not followed</strong> — a real action in the session contradicts the rule. The evidence quotes it.</li>
|
|
187
|
-
<li><strong>
|
|
246
|
+
<li><strong>Couldn't tell</strong> — the check ran and the evidence was ambiguous. A genuine gap.</li>
|
|
247
|
+
<li><strong>Needs your judgment</strong> — this rule never had a mechanical answer ("surface bad news first"). Not a pass, not a failure, and not a shortcoming of the check: it is the part that was always a person's call.</li>
|
|
188
248
|
</ul>
|
|
189
249
|
<h2>What this report does not establish</h2>
|
|
190
250
|
<ul>
|
|
@@ -44,11 +44,32 @@ function ruleLabel(r, results) {
|
|
|
44
44
|
const collides = results.filter((other) => other.ruleId === r.ruleId).length > 1;
|
|
45
45
|
return collides ? `Rule ${r.ruleId} (${r.ruleSource}) — ${r.ruleTitle}` : `Rule ${r.ruleId} — ${r.ruleTitle}`;
|
|
46
46
|
}
|
|
47
|
+
/**
|
|
48
|
+
* Separates the two very different things that were both being reported
|
|
49
|
+
* as "unclear":
|
|
50
|
+
*
|
|
51
|
+
* - couldn't tell — the tool looked and the evidence was ambiguous.
|
|
52
|
+
* - needs you — a judgment call that never had a mechanical
|
|
53
|
+
* answer ("surface bad news first").
|
|
54
|
+
*
|
|
55
|
+
* Collapsing them made the tool look like it failed on most rules. It
|
|
56
|
+
* doesn't: across 40 real rules files, 52.2% of the actual rules people
|
|
57
|
+
* write are judgment calls. Those aren't the tool falling short, they're
|
|
58
|
+
* the half of the work that was always a human's. Naming that honestly is
|
|
59
|
+
* the difference between a report that reads as broken and one that reads
|
|
60
|
+
* as a division of labour.
|
|
61
|
+
*/
|
|
47
62
|
function summaryLine(results) {
|
|
48
63
|
const pass = results.filter((r) => r.status === "PASS").length;
|
|
49
64
|
const fail = results.filter((r) => r.status === "FAIL").length;
|
|
50
|
-
const
|
|
51
|
-
|
|
65
|
+
const needsHuman = results.filter((r) => r.status === "UNCLEAR" && r.needsHuman).length;
|
|
66
|
+
const couldntTell = results.filter((r) => r.status === "UNCLEAR" && !r.needsHuman).length;
|
|
67
|
+
const parts = [`${pass} followed`, `${fail} not followed`];
|
|
68
|
+
if (couldntTell > 0)
|
|
69
|
+
parts.push(`${couldntTell} couldn't tell`);
|
|
70
|
+
if (needsHuman > 0)
|
|
71
|
+
parts.push(`${needsHuman} need your judgment`);
|
|
72
|
+
return parts.join(" · ");
|
|
52
73
|
}
|
|
53
74
|
export function generateReport(results, meta) {
|
|
54
75
|
const clean = results.map(sanitize);
|
package/dist/types.d.ts
CHANGED
|
@@ -32,4 +32,20 @@ export interface CheckResult {
|
|
|
32
32
|
ruleSource: "global" | "project";
|
|
33
33
|
status: CheckStatus;
|
|
34
34
|
evidence: string;
|
|
35
|
+
/**
|
|
36
|
+
* True when this rule was never mechanically answerable — a judgment
|
|
37
|
+
* call like "surface bad news first", which has no command to inspect.
|
|
38
|
+
*
|
|
39
|
+
* Exists because collapsing these into a single UNCLEAR count made the
|
|
40
|
+
* tool look broken. Measured across 40 real rules files: of the actual
|
|
41
|
+
* rules people write, 47.8% are mechanically answerable and 52.2% are
|
|
42
|
+
* judgment calls. Reporting "1 pass · 0 fail · 14 unclear" reads as
|
|
43
|
+
* fourteen failures, when most of those were a human's call from the
|
|
44
|
+
* start and the tool is working exactly as intended.
|
|
45
|
+
*
|
|
46
|
+
* "I could not determine this" and "this was always yours to decide"
|
|
47
|
+
* are different statements, and a tool about honest reporting should
|
|
48
|
+
* not blur them.
|
|
49
|
+
*/
|
|
50
|
+
needsHuman?: boolean;
|
|
35
51
|
}
|