rulereceipt 0.1.67 → 0.1.69
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/browser/evaluateBrowser.js +3 -1
- package/dist/checks/claimEvidence.js +3 -1
- package/dist/checks/classify.js +27 -1
- package/dist/checks/codeContent.js +22 -2
- package/dist/checks/userAsked.d.ts +15 -0
- package/dist/checks/userAsked.js +122 -0
- package/dist/evaluate.js +3 -1
- package/package.json +2 -2
|
@@ -11,6 +11,7 @@ import { runEmojiChecks } from "../checks/emojiOutput.js";
|
|
|
11
11
|
import { runAttributionChecks } from "../checks/attribution.js";
|
|
12
12
|
import { runApprovalGateChecks } from "../checks/approvalGate.js";
|
|
13
13
|
import { touchedPaths, ruleWasLoaded } from "../checks/pathScope.js";
|
|
14
|
+
import { downgradeUserAsked } from "../checks/userAsked.js";
|
|
14
15
|
/**
|
|
15
16
|
* Evaluate a session against rules ENTIRELY in the browser — the same checkers
|
|
16
17
|
* the CLI uses, run on a file the user dropped, with nothing leaving the page.
|
|
@@ -62,7 +63,8 @@ export function evaluateBrowserSession(rulesText, sessionText) {
|
|
|
62
63
|
needsHuman: true,
|
|
63
64
|
evidence: "",
|
|
64
65
|
}));
|
|
65
|
-
|
|
66
|
+
const structural = downgradeUserAsked(deterministicResults, classifications, events);
|
|
67
|
+
return [...structural, ...judgmentResults, ...scopeResults];
|
|
66
68
|
}
|
|
67
69
|
/** The whole client-side session check: parse, evaluate, and count. */
|
|
68
70
|
export function checkSessionInBrowser(rulesText, sessionText) {
|
|
@@ -83,7 +83,9 @@ const ACTION_CLAIMS = [
|
|
|
83
83
|
// Idioms that borrow "pushed" but aren't a git push. "pushed the fix/code/
|
|
84
84
|
// changes/branch to <remote>" stays a claim; effort/figurative senses do
|
|
85
85
|
// not (finding #6, 2026-09-26).
|
|
86
|
-
|
|
86
|
+
// "pushed nothing/none" is a statement that NO push happened — the opposite
|
|
87
|
+
// of a push claim (finding 2026-09-29, false-accusation corpus run).
|
|
88
|
+
exclude: /\bpushed\s+(?:back|for|through|forward|ahead|past|hard|on|nothing|none|myself|ourselves|yourself|themselves|the\s+(?:boundar|button|envelope|limit|deadline|pace)|to\s+(?:get|finish|complete|ship|meet|hit|make|wrap|move))/i,
|
|
87
89
|
command: /\bgit\s+push\b/i,
|
|
88
90
|
},
|
|
89
91
|
{
|
package/dist/checks/classify.js
CHANGED
|
@@ -312,6 +312,28 @@ function isContentToken(literal) {
|
|
|
312
312
|
return false;
|
|
313
313
|
return (literal.match(/[A-Za-z0-9]/g) ?? []).length >= 3;
|
|
314
314
|
}
|
|
315
|
+
/**
|
|
316
|
+
* A file/directory/route token, as opposed to a code construct or an import
|
|
317
|
+
* specifier. It names a PATH, so searching file CONTENT for it (codeContent's
|
|
318
|
+
* job) matches every import, changelog line, package.json entry or prose that
|
|
319
|
+
* merely MENTIONS the path — a "found `dist/` written into a file" false
|
|
320
|
+
* accusation. Found in the false-accusation corpus run 2026-09-29: ~30 distinct
|
|
321
|
+
* FAIL texts were exactly this (`dist/`, `src/`, `README.md`, `AGENTS.md`,
|
|
322
|
+
* `./types`, `/status`, …). These belong to fileLifecycle (a path mutation) or,
|
|
323
|
+
* absent a mutation verb, to the deterministic checker, which reports UNCLEAR
|
|
324
|
+
* rather than fabricating a FAIL.
|
|
325
|
+
*
|
|
326
|
+
* Deliberately NARROW so real import specifiers stay checkable: `lucide-react`
|
|
327
|
+
* and `@scope/pkg` are NOT file tokens (no leading `./` or `/`, no trailing
|
|
328
|
+
* `/`, no file extension), so a "never import X" rule keeps working.
|
|
329
|
+
*/
|
|
330
|
+
const FILE_EXTENSION = /\.(json|ya?ml|toml|md|mdx|env|lock|ini|cfg|conf|xml|txt|js|jsx|mjs|cjs|ts|tsx|py|rb|go|rs|sh|sql|css|scss|html)$/i;
|
|
331
|
+
function looksLikeFilePathToken(literal) {
|
|
332
|
+
return (/\/$/.test(literal) || // trailing slash: a directory ("dist/", "src/")
|
|
333
|
+
/^(?:\.{1,2}\/|\/)/.test(literal) || // leading ./ ../ or /: a path or route
|
|
334
|
+
FILE_EXTENSION.test(literal) // a bare filename with a known extension
|
|
335
|
+
);
|
|
336
|
+
}
|
|
315
337
|
// Catches rules like "add tests for every change" or "every new function
|
|
316
338
|
// needs a test" - no literal backtick token to pattern-match, so without
|
|
317
339
|
// this they'd fall all the way through to judgment (an LLM call) even
|
|
@@ -743,7 +765,11 @@ export function classifyRule(rule) {
|
|
|
743
765
|
// only, so this is an action not a mention. Plain words and shell commands
|
|
744
766
|
// are excluded by isContentToken; protected files already routed above.
|
|
745
767
|
if (polarity === "forbid") {
|
|
746
|
-
|
|
768
|
+
// A file/dir/route token is a path reference, not code to search for inside
|
|
769
|
+
// file content — routing it to codeContent produces "found `dist/` written
|
|
770
|
+
// into a file" on any mention. Excluded here; it falls through to the
|
|
771
|
+
// deterministic checker (UNCLEAR, never a fabricated FAIL).
|
|
772
|
+
const contentTokens = [...patterns].filter((p) => isContentToken(p) && !looksLikeFilePathToken(p));
|
|
747
773
|
if (contentTokens.length > 0) {
|
|
748
774
|
return { kind: "codeContent", rule, patterns: contentTokens, polarity, polarityInferred };
|
|
749
775
|
}
|
|
@@ -47,8 +47,28 @@ function editedContentFromEvent(event) {
|
|
|
47
47
|
*/
|
|
48
48
|
function containsCall(content, pattern) {
|
|
49
49
|
const leadsWithIdentifier = /^[A-Za-z0-9_$]/.test(pattern);
|
|
50
|
-
if (!leadsWithIdentifier)
|
|
51
|
-
|
|
50
|
+
if (!leadsWithIdentifier) {
|
|
51
|
+
// A CALL like `.forEach(` legitimately follows an object (`arr.forEach()`),
|
|
52
|
+
// so member access before it is fine — bare containment.
|
|
53
|
+
if (pattern.endsWith("("))
|
|
54
|
+
return content.includes(pattern);
|
|
55
|
+
// A punct-leading NON-call token (a dotfile/extension like `.env`, `.log`)
|
|
56
|
+
// sitting right after an identifier is a property access or the tail of a
|
|
57
|
+
// longer token, not the token itself: `.env` inside `process.env` is not
|
|
58
|
+
// the .env file. Require a non-identifier char (or the start) before it.
|
|
59
|
+
// Found in the false-accusation corpus run 2026-09-29 — a `.env` rule
|
|
60
|
+
// FAILed every file using `process.env`.
|
|
61
|
+
let fromPunct = 0;
|
|
62
|
+
for (;;) {
|
|
63
|
+
const at = content.indexOf(pattern, fromPunct);
|
|
64
|
+
if (at === -1)
|
|
65
|
+
return false;
|
|
66
|
+
const before = at === 0 ? "" : content[at - 1];
|
|
67
|
+
if (!/[A-Za-z0-9_$]/.test(before))
|
|
68
|
+
return true;
|
|
69
|
+
fromPunct = at + 1;
|
|
70
|
+
}
|
|
71
|
+
}
|
|
52
72
|
let from = 0;
|
|
53
73
|
for (;;) {
|
|
54
74
|
const at = content.indexOf(pattern, from);
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
import type { TranscriptEvent, CheckResult } from "../types.js";
|
|
2
|
+
import type { Classification } from "./classify.js";
|
|
3
|
+
/**
|
|
4
|
+
* The user's quote if a user message in this session explicitly instructed one
|
|
5
|
+
* of the rule's forbidden subjects; otherwise null. Only user text is read
|
|
6
|
+
* (never the agent's own words, never tool output).
|
|
7
|
+
*/
|
|
8
|
+
export declare function userAskedFor(events: TranscriptEvent[], subjects: string[]): string | null;
|
|
9
|
+
/**
|
|
10
|
+
* Downgrades a structured FAIL to "can't tell" when the user explicitly asked
|
|
11
|
+
* for the rule's forbidden subject. Only ever FAIL -> UNCLEAR, never the
|
|
12
|
+
* reverse — so it can only remove a false accusation. Shared by the CLI and the
|
|
13
|
+
* browser evaluator so the two never disagree about it.
|
|
14
|
+
*/
|
|
15
|
+
export declare function downgradeUserAsked(results: CheckResult[], classifications: Classification[], events: TranscriptEvent[]): CheckResult[];
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* "You asked for it" — for every structured rule, not just push/commit/pr.
|
|
3
|
+
*
|
|
4
|
+
* If the agent did the thing a rule forbids, but the USER explicitly told it to
|
|
5
|
+
* in this session, that is the user overriding their own rule, not the agent
|
|
6
|
+
* breaking it. The report must not accuse the agent of it. approvalGate already
|
|
7
|
+
* does this for push/commit/pr/delete; this generalises it to the other
|
|
8
|
+
* structured checkers (a forbidden branch, file, or code token).
|
|
9
|
+
*
|
|
10
|
+
* It ONLY ever downgrades a FAIL to "can't tell" (with the user's quote) — it
|
|
11
|
+
* can never create a FAIL — so, like every other moat fix, it can only remove
|
|
12
|
+
* a false accusation. It is deliberately conservative to protect DETECTION: the
|
|
13
|
+
* user's message must name the rule's specific subject, not be negated, and not
|
|
14
|
+
* be a question about it. A vague "ship it" is not "push to `main`", so a real
|
|
15
|
+
* violation there still stands.
|
|
16
|
+
*/
|
|
17
|
+
/** The forbidden thing was named, not negated ("don't edit .env" is not a yes). */
|
|
18
|
+
function negatedBefore(text, at) {
|
|
19
|
+
const before = text.slice(Math.max(0, at - 32), at);
|
|
20
|
+
return /\b(?:don'?t|do\s+not|never|no|not\s+yet|without|avoid|stop|hold\s+off(?:\s+on)?|refrain\s+from)\b[\s\w'-]{0,16}$/i.test(before);
|
|
21
|
+
}
|
|
22
|
+
/** A clause that only ASKS about the subject ("should we edit .env?") is not an instruction. */
|
|
23
|
+
function isQuestionClause(clause) {
|
|
24
|
+
return /\?/.test(clause) || /^\s*(?:should|could|would|why|what|whether|is|are|do|does|can|shall|may)\b/i.test(clause.trim());
|
|
25
|
+
}
|
|
26
|
+
function escapeRegex(s) {
|
|
27
|
+
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* A regex that matches the rule's subject as a whole token, so `.env` does not
|
|
31
|
+
* match inside `.env.example` and `main` does not match inside `maintain`.
|
|
32
|
+
* A trailing `(` on a code token (`console.log(`) is dropped for matching.
|
|
33
|
+
*/
|
|
34
|
+
function subjectMatcher(subject) {
|
|
35
|
+
const bare = subject.replace(/\($/, "").trim();
|
|
36
|
+
const leadBoundary = /^[\w]/.test(bare) ? "(?<![\\w-])" : "";
|
|
37
|
+
const trailBoundary = /[\w]$/.test(bare) ? "(?![\\w.-])" : "(?![\\w.-])";
|
|
38
|
+
return new RegExp(leadBoundary + escapeRegex(bare) + trailBoundary, "i");
|
|
39
|
+
}
|
|
40
|
+
/**
|
|
41
|
+
* The user's quote if a user message in this session explicitly instructed one
|
|
42
|
+
* of the rule's forbidden subjects; otherwise null. Only user text is read
|
|
43
|
+
* (never the agent's own words, never tool output).
|
|
44
|
+
*/
|
|
45
|
+
export function userAskedFor(events, subjects) {
|
|
46
|
+
const usable = subjects.map((s) => s.replace(/\($/, "").trim()).filter((s) => s.length >= 2);
|
|
47
|
+
if (usable.length === 0)
|
|
48
|
+
return null;
|
|
49
|
+
for (const e of events) {
|
|
50
|
+
if (e.kind !== "text" || e.role !== "user")
|
|
51
|
+
continue;
|
|
52
|
+
// Look clause by clause, so a negation or question in one sentence doesn't
|
|
53
|
+
// wrongly clear (or wrongly count) another.
|
|
54
|
+
for (const clause of e.text.split(/(?<=[.!?\n])\s+/)) {
|
|
55
|
+
if (isQuestionClause(clause))
|
|
56
|
+
continue;
|
|
57
|
+
for (const subject of usable) {
|
|
58
|
+
const m = clause.match(subjectMatcher(subject));
|
|
59
|
+
if (!m || m.index === undefined)
|
|
60
|
+
continue;
|
|
61
|
+
if (negatedBefore(clause, m.index))
|
|
62
|
+
continue;
|
|
63
|
+
return clause.replace(/\s+/g, " ").trim().slice(0, 160);
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
return null;
|
|
68
|
+
}
|
|
69
|
+
/** The forbidden subject(s) of a structured classification, or null if it has none we can name. */
|
|
70
|
+
function subjectsOf(c) {
|
|
71
|
+
switch (c.kind) {
|
|
72
|
+
case "gitBranchPolicy":
|
|
73
|
+
return [c.branchName];
|
|
74
|
+
case "fileLifecycle":
|
|
75
|
+
return [c.filePath];
|
|
76
|
+
case "codeContent":
|
|
77
|
+
return c.patterns;
|
|
78
|
+
case "emojiOutput":
|
|
79
|
+
return ["emoji"];
|
|
80
|
+
default:
|
|
81
|
+
// attribution deliberately excluded: an AI-attribution trailer is never
|
|
82
|
+
// something a user meaningfully instructs the agent to author as its own,
|
|
83
|
+
// and it is the project's own hard rule. approvalGate handles its own asks.
|
|
84
|
+
return null;
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
function ruleKey(source, id, title) {
|
|
88
|
+
return `${source}\u0000${id}\u0000${title}`;
|
|
89
|
+
}
|
|
90
|
+
/**
|
|
91
|
+
* Downgrades a structured FAIL to "can't tell" when the user explicitly asked
|
|
92
|
+
* for the rule's forbidden subject. Only ever FAIL -> UNCLEAR, never the
|
|
93
|
+
* reverse — so it can only remove a false accusation. Shared by the CLI and the
|
|
94
|
+
* browser evaluator so the two never disagree about it.
|
|
95
|
+
*/
|
|
96
|
+
export function downgradeUserAsked(results, classifications, events) {
|
|
97
|
+
const subjects = new Map();
|
|
98
|
+
for (const c of classifications) {
|
|
99
|
+
const s = subjectsOf(c);
|
|
100
|
+
if (s && s.length > 0)
|
|
101
|
+
subjects.set(ruleKey(c.rule.source, c.rule.id, c.rule.title), s);
|
|
102
|
+
}
|
|
103
|
+
if (subjects.size === 0)
|
|
104
|
+
return results;
|
|
105
|
+
return results.map((r) => {
|
|
106
|
+
if (r.status !== "FAIL")
|
|
107
|
+
return r;
|
|
108
|
+
const s = subjects.get(ruleKey(r.ruleSource, r.ruleId, r.ruleTitle));
|
|
109
|
+
if (!s)
|
|
110
|
+
return r;
|
|
111
|
+
const quote = userAskedFor(events, s);
|
|
112
|
+
if (!quote)
|
|
113
|
+
return r;
|
|
114
|
+
return {
|
|
115
|
+
...r,
|
|
116
|
+
status: "UNCLEAR",
|
|
117
|
+
outcome: "inconclusive",
|
|
118
|
+
reason: "user_asked",
|
|
119
|
+
evidence: `the agent did this, but you asked for it in this session: "${quote}" — so this is you overriding your own rule, not the agent breaking it. Original finding: ${r.evidence}`,
|
|
120
|
+
};
|
|
121
|
+
});
|
|
122
|
+
}
|
package/dist/evaluate.js
CHANGED
|
@@ -10,6 +10,7 @@ import { runAttributionChecks } from "./checks/attribution.js";
|
|
|
10
10
|
import { runApprovalGateChecks } from "./checks/approvalGate.js";
|
|
11
11
|
import { runJudgmentChecks } from "./checks/judgmentChecks.js";
|
|
12
12
|
import { touchedPaths, ruleWasLoaded } from "./checks/pathScope.js";
|
|
13
|
+
import { downgradeUserAsked } from "./checks/userAsked.js";
|
|
13
14
|
import { readFileSync } from "node:fs";
|
|
14
15
|
import { homedir } from "node:os";
|
|
15
16
|
import { join } from "node:path";
|
|
@@ -98,7 +99,8 @@ export async function evaluateSession(cwd, rules, events, llm, needsLlmResult) {
|
|
|
98
99
|
const judgmentResults = llm
|
|
99
100
|
? await runJudgmentChecks(judgment, events)
|
|
100
101
|
: judgment.map(({ rule }) => needsLlmResult(rule));
|
|
101
|
-
const
|
|
102
|
+
const structuralResults = downgradeUserAsked(deterministicResults, classifications, events);
|
|
103
|
+
const results = [...structuralResults, ...judgmentResults, ...scopeResults, ...future.map(futureResult)];
|
|
102
104
|
return {
|
|
103
105
|
results: attachSourceLocation(results, rules),
|
|
104
106
|
notARule: of("notARule"),
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "rulereceipt",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.69",
|
|
4
4
|
"description": "Checks whether your AI coding agent followed your rules, with evidence. Works with Claude Code (Codex in testing); reads CLAUDE.md, AGENTS.md, Cursor, Copilot and Windsurf rules.",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
@@ -48,7 +48,7 @@
|
|
|
48
48
|
"verify": "npm run lint && npm run typecheck && npm test"
|
|
49
49
|
},
|
|
50
50
|
"engines": {
|
|
51
|
-
"node": ">=
|
|
51
|
+
"node": ">=20"
|
|
52
52
|
},
|
|
53
53
|
"license": "SEE LICENSE IN LICENSE",
|
|
54
54
|
"dependencies": {
|