rulereceipt 0.1.53 → 0.1.54
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -7
- package/dist/adapters/codex.d.ts +5 -0
- package/dist/adapters/codex.js +263 -0
- package/dist/adapters/index.d.ts +82 -0
- package/dist/adapters/index.js +133 -0
- package/dist/checks/attribution.js +29 -8
- package/dist/checks/claimEvidence.js +52 -3
- package/dist/checks/codeContent.js +5 -1
- package/dist/checks/deterministicChecks.js +56 -4
- package/dist/checks/emojiOutput.js +29 -10
- package/dist/checks/fileLifecycle.js +29 -15
- package/dist/checks/gitBranchPolicy.js +86 -14
- package/dist/checks/ifEditThenTest.js +7 -1
- package/dist/checks/proposedAction.js +20 -32
- package/dist/checks/shellCommand.d.ts +22 -0
- package/dist/checks/shellCommand.js +67 -4
- package/dist/checks/testCommands.js +12 -3
- package/dist/cli.js +20 -5
- package/dist/parsers/readClaudeMd.d.ts +0 -17
- package/dist/parsers/readClaudeMd.js +20 -1
- package/dist/parsers/readMemory.d.ts +2 -0
- package/dist/parsers/readMemory.js +99 -0
- package/dist/report/complianceReport.js +7 -4
- package/dist/rules.js +35 -5
- package/package.json +7 -2
|
@@ -48,7 +48,10 @@ const SUCCESS_CLAIM = /\b(?:all\s+)?(?:the\s+)?tests?(?:\s+suite)?\s+(?:are|is|n
|
|
|
48
48
|
* sits in, not the whole message — a paragraph that says "one is failing"
|
|
49
49
|
* elsewhere should not excuse a false claim made in its own sentence.
|
|
50
50
|
*/
|
|
51
|
-
|
|
51
|
+
// `getting`/`means`/`goal` added 2026-09-26: "Getting tests passing is the
|
|
52
|
+
// last step" and "All tests passing means the refactor is complete" are goal
|
|
53
|
+
// framings, not a report that the tests currently pass (finding #5).
|
|
54
|
+
const NOT_A_CLAIM = /(?:\b(?:if|unless|once|when|after|before|until|should|would|will|going to|i'?ll|let'?s|need to|make sure|ensure|hope|expect|check (?:if|whether)|verify (?:that|if)|not|cannot|getting|means|goal|fail(?:s|ing|ed)?|red|broken)\b|\w+n['\u2019]t\b)/i;
|
|
52
55
|
/**
|
|
53
56
|
* Actions the session can claim to have performed, and the command that
|
|
54
57
|
* would prove it.
|
|
@@ -77,7 +80,10 @@ const ACTION_CLAIMS = [
|
|
|
77
80
|
{
|
|
78
81
|
label: "git push",
|
|
79
82
|
claim: /\b(?:i|we)(?:'ve|\u2019ve| have| had)?\s+(?:\w+ly\s+|just\s+|already\s+|then\s+|also\s+|now\s+)*pushed\b/i,
|
|
80
|
-
|
|
83
|
+
// Idioms that borrow "pushed" but aren't a git push. "pushed the fix/code/
|
|
84
|
+
// changes/branch to <remote>" stays a claim; effort/figurative senses do
|
|
85
|
+
// not (finding #6, 2026-09-26).
|
|
86
|
+
exclude: /\bpushed\s+(?:back|for|through|forward|ahead|past|hard|on|myself|ourselves|yourself|themselves|the\s+(?:boundar|button|envelope|limit|deadline|pace)|to\s+(?:get|finish|complete|ship|meet|hit|make|wrap|move))/i,
|
|
81
87
|
command: /\bgit\s+push\b/i,
|
|
82
88
|
},
|
|
83
89
|
{
|
|
@@ -111,7 +117,11 @@ const ACTION_CLAIMS = [
|
|
|
111
117
|
* something else entirely is still beyond it.
|
|
112
118
|
*/
|
|
113
119
|
label: "read of a source",
|
|
114
|
-
|
|
120
|
+
// A bare "status" header no longer counts on digits alone — "Status: 3 of
|
|
121
|
+
// 5 tasks done" / "STATUS: 200" are ordinary status lines, not a claim of
|
|
122
|
+
// having read a source (finding #7, 2026-09-26). "PAGES READ: <n>" and
|
|
123
|
+
// "STATUS: READ IN FULL" are the real provenance forms and still count.
|
|
124
|
+
claim: /\b(?:i|we)(?:'ve|’ve| have| had)?\s+(?:\w+ly\s+|just\s+|already\s+|then\s+|also\s+|now\s+)*read\b|^\s*pages?\s+read\s*:\s*(?:[\d\s,-]+|read\s+in\s+full)|^\s*status\s*:\s*read\s+in\s+full|\bread\s+in\s+full\b|\bconfirmed\s+at\s+source\b/im,
|
|
115
125
|
exclude: /\b(?:will|going\s+to|need\s+to|should|next|plan\s+to|about\s+to|let\s+me|i'?ll|we'?ll)\s+(?:\w+\s+){0,3}read\b/i,
|
|
116
126
|
command: /\b(?:cat|head|tail|less|more|bat|nl|strings|pdftotext|xxd|od)\b/i,
|
|
117
127
|
},
|
|
@@ -216,6 +226,22 @@ function unclear(rule, evidence) {
|
|
|
216
226
|
evidence,
|
|
217
227
|
};
|
|
218
228
|
}
|
|
229
|
+
/**
|
|
230
|
+
* A run of the WHOLE suite, not a subset. A scoped run names a file/path or a
|
|
231
|
+
* `-- <filter>` positional. Used so a subset pass (`npm test -- frontend`)
|
|
232
|
+
* does not clear an earlier failure of a different scope (`… -- backend`) — a
|
|
233
|
+
* broad "all passing" claim over a partial re-run is unbacked, not proven
|
|
234
|
+
* (finding #8, 2026-09-26).
|
|
235
|
+
*/
|
|
236
|
+
function isFullSuiteRun(command) {
|
|
237
|
+
if (/\s--\s+[^\s-]/.test(command))
|
|
238
|
+
return false;
|
|
239
|
+
if (/\b(?:pytest|jest|vitest|mocha|phpunit|rspec)\s+[^\s-]\S*\.\w+/.test(command))
|
|
240
|
+
return false;
|
|
241
|
+
if (/\s\S*\.(?:test|spec)\.\w+/.test(command))
|
|
242
|
+
return false;
|
|
243
|
+
return true;
|
|
244
|
+
}
|
|
219
245
|
/**
|
|
220
246
|
* Walks the session in order, tracking the state of the last test run, and
|
|
221
247
|
* tests every assistant claim against the state at the moment it was made.
|
|
@@ -240,6 +266,10 @@ export function runClaimEvidenceChecks(classifications, events) {
|
|
|
240
266
|
let uncertain = null;
|
|
241
267
|
let unreadable = null;
|
|
242
268
|
let backed = null;
|
|
269
|
+
// A failing run whose failure has NOT been cleared by a full-suite (or
|
|
270
|
+
// same-scope) passing re-run. A subset pass afterward leaves this standing.
|
|
271
|
+
let unresolvedFailure = null;
|
|
272
|
+
let partialPass = null;
|
|
243
273
|
for (const event of events) {
|
|
244
274
|
// A read through Read/Grep/Glob never reaches the shell, so it has to be
|
|
245
275
|
// recorded here rather than by matching command text.
|
|
@@ -293,6 +323,15 @@ export function runClaimEvidenceChecks(classifications, events) {
|
|
|
293
323
|
};
|
|
294
324
|
pendingRuns.delete(resultId);
|
|
295
325
|
unknownSinceRed = null; // a recognised run supersedes anything before it
|
|
326
|
+
// Track whether a failure is still standing. A full-suite (or
|
|
327
|
+
// same-command) pass clears it; a subset pass does not.
|
|
328
|
+
if (lastRun.outcomeReadable && lastRun.failed) {
|
|
329
|
+
unresolvedFailure = pendingRun;
|
|
330
|
+
}
|
|
331
|
+
else if (lastRun.outcomeReadable && !lastRun.failed && unresolvedFailure !== null) {
|
|
332
|
+
if (isFullSuiteRun(pendingRun) || pendingRun === unresolvedFailure)
|
|
333
|
+
unresolvedFailure = null;
|
|
334
|
+
}
|
|
296
335
|
}
|
|
297
336
|
continue;
|
|
298
337
|
}
|
|
@@ -340,6 +379,12 @@ export function runClaimEvidenceChecks(classifications, events) {
|
|
|
340
379
|
else if (lastRun.failed && contradiction === null) {
|
|
341
380
|
contradiction = { claim: sentence.trim(), run: lastRun };
|
|
342
381
|
}
|
|
382
|
+
else if (!lastRun.failed && unresolvedFailure !== null && partialPass === null) {
|
|
383
|
+
// The last run passed, but it was a subset — an earlier failure of a
|
|
384
|
+
// different scope was never re-verified. Can't back a broad claim from
|
|
385
|
+
// a partial pass, and can't accuse either. Report the gap.
|
|
386
|
+
partialPass = { claim: sentence.trim(), failing: unresolvedFailure };
|
|
387
|
+
}
|
|
343
388
|
else if (!lastRun.failed && backed === null) {
|
|
344
389
|
backed = { claim: sentence.trim(), run: lastRun };
|
|
345
390
|
}
|
|
@@ -367,6 +412,10 @@ export function runClaimEvidenceChecks(classifications, events) {
|
|
|
367
412
|
return unclear(rule, `the session stated: "${uncertain.claim}" after a failing test run, but \`${uncertain.script}\` ` +
|
|
368
413
|
`ran in between and this tool cannot tell whether that re-ran the suite — a human has to look`);
|
|
369
414
|
}
|
|
415
|
+
if (partialPass && !contradiction) {
|
|
416
|
+
return unclear(rule, `the session stated: "${partialPass.claim}", but the earlier failing run \`${short(partialPass.failing)}\` ` +
|
|
417
|
+
`was only re-run in part — no full-suite pass covered it, so this claim isn't backed (nor disproven) here`);
|
|
418
|
+
}
|
|
370
419
|
if (contradiction) {
|
|
371
420
|
return {
|
|
372
421
|
ruleId: rule.id,
|
|
@@ -55,7 +55,11 @@ function containsCall(content, pattern) {
|
|
|
55
55
|
if (at === -1)
|
|
56
56
|
return false;
|
|
57
57
|
const before = at === 0 ? "" : content[at - 1];
|
|
58
|
-
|
|
58
|
+
// A `.` before the pattern is a member-access SEPARATOR, not identifier
|
|
59
|
+
// continuation — `analytics.track(` is a real call to `track(`. So `.` is
|
|
60
|
+
// NOT in the disqualifying class (fixed 2026-09-26); `_` still is, so
|
|
61
|
+
// `_metar_fetch(` does not match `fetch(`.
|
|
62
|
+
if (!/[A-Za-z0-9_$]/.test(before))
|
|
59
63
|
return true;
|
|
60
64
|
from = at + 1;
|
|
61
65
|
}
|
|
@@ -1,3 +1,39 @@
|
|
|
1
|
+
import { segments, leadingCommand, withoutCommitMessage } from "./shellCommand.js";
|
|
2
|
+
/**
|
|
3
|
+
* Commands that only READ/print/search their arguments — a required pattern
|
|
4
|
+
* appearing as an argument to one of these is being mentioned, not run.
|
|
5
|
+
*/
|
|
6
|
+
const MENTION_ONLY_CMDS = new Set([
|
|
7
|
+
"grep", "rg", "ag", "ack", "egrep", "fgrep", "echo", "printf", "cat", "bat",
|
|
8
|
+
"head", "tail", "less", "more", "ls", "find", "fd", "jq", "yq", "cut", "tr",
|
|
9
|
+
"sort", "uniq", "diff", "comm", "wc", "which", "type", "file", "stat", "man",
|
|
10
|
+
]);
|
|
11
|
+
/**
|
|
12
|
+
* Does this event actually PERFORM the required pattern, rather than mention
|
|
13
|
+
* it? A required action reported satisfied on a mention (assistant text, a
|
|
14
|
+
* grep/echo, a commit-message reference) is the mirror of the forbid path's
|
|
15
|
+
* mention-vs-action confusion, and it PASSes a rule that was never followed
|
|
16
|
+
* (finding 2026-09-26). Text and tool_result never count; a Bash command
|
|
17
|
+
* counts only when the pattern sits in a segment that is actually executed.
|
|
18
|
+
*/
|
|
19
|
+
function requirePerformedBy(event, pattern) {
|
|
20
|
+
if (event.kind !== "tool_use")
|
|
21
|
+
return false;
|
|
22
|
+
if (event.toolName === "Bash") {
|
|
23
|
+
const input = event.input;
|
|
24
|
+
const command = input && typeof input.command === "string" ? input.command : "";
|
|
25
|
+
for (const seg of segments(command)) {
|
|
26
|
+
if (MENTION_ONLY_CMDS.has(leadingCommand(seg)))
|
|
27
|
+
continue;
|
|
28
|
+
if (matchesPattern(canonicalise(withoutCommitMessage(seg)), pattern))
|
|
29
|
+
return true;
|
|
30
|
+
}
|
|
31
|
+
return false;
|
|
32
|
+
}
|
|
33
|
+
// A non-Bash tool_use (Write/Edit/…) that carries the pattern in its input
|
|
34
|
+
// genuinely produced it (e.g. `Closes #N` written into a PR body).
|
|
35
|
+
return matchesPattern(searchHaystack(event), pattern);
|
|
36
|
+
}
|
|
1
37
|
/**
|
|
2
38
|
* Real false-positive found 2026-08-30 on an actual complex session: a
|
|
3
39
|
* "no debug print() statements" rule failed because the agent ran a grep
|
|
@@ -174,14 +210,30 @@ export function runDeterministicChecks(classifications, events) {
|
|
|
174
210
|
// distinguish "should have run this and didn't" from "this session
|
|
175
211
|
// never needed to" — so a required-but-absent pattern reports
|
|
176
212
|
// UNCLEAR, never a fabricated FAIL or a fabricated PASS.
|
|
177
|
-
|
|
178
|
-
|
|
213
|
+
// A match is only a PASS when the pattern was actually PERFORMED, not
|
|
214
|
+
// merely mentioned (in assistant text, a grep/echo, or a commit
|
|
215
|
+
// message) — otherwise a required action reads as done when it wasn't.
|
|
216
|
+
let performedEvent;
|
|
217
|
+
let performedPattern;
|
|
218
|
+
for (const event of events) {
|
|
219
|
+
for (const pattern of patterns) {
|
|
220
|
+
if (requirePerformedBy(event, pattern)) {
|
|
221
|
+
performedEvent = event;
|
|
222
|
+
performedPattern = pattern;
|
|
223
|
+
break;
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
if (performedEvent)
|
|
227
|
+
break;
|
|
228
|
+
}
|
|
229
|
+
if (performedEvent && performedPattern) {
|
|
230
|
+
const haystack = eventSearchText(performedEvent);
|
|
179
231
|
return {
|
|
180
232
|
ruleId: rule.id,
|
|
181
233
|
ruleTitle: rule.title,
|
|
182
234
|
ruleSource: rule.source,
|
|
183
235
|
status: "PASS",
|
|
184
|
-
evidence: `found required pattern "${
|
|
236
|
+
evidence: `found required pattern "${performedPattern}" in a ${performedEvent.kind === "tool_use" ? performedEvent.toolName + " call" : performedEvent.kind}: ${haystack.slice(0, 160)}`,
|
|
185
237
|
};
|
|
186
238
|
}
|
|
187
239
|
return {
|
|
@@ -189,7 +241,7 @@ export function runDeterministicChecks(classifications, events) {
|
|
|
189
241
|
ruleTitle: rule.title,
|
|
190
242
|
ruleSource: rule.source,
|
|
191
243
|
status: "UNCLEAR",
|
|
192
|
-
evidence: `required pattern ${patterns.map((p) => `"${p}"`).join(" or ")}
|
|
244
|
+
evidence: `required pattern ${patterns.map((p) => `"${p}"`).join(" or ")} was not actually run this session — can't tell if the rule didn't apply, or applied and was skipped`,
|
|
193
245
|
};
|
|
194
246
|
}
|
|
195
247
|
catch (err) {
|
|
@@ -30,27 +30,46 @@ import { violation } from "../types.js";
|
|
|
30
30
|
const DEFAULT_EMOJI = /\p{Emoji_Presentation}/u;
|
|
31
31
|
const PICTOGRAPHIC = /\p{Extended_Pictographic}/u;
|
|
32
32
|
const REGIONAL_INDICATOR = /[\u{1F1E6}-\u{1F1FF}]/u;
|
|
33
|
-
const KEYCAP = /[0-9#*]\u{FE0F}?\u{20E3}/u;
|
|
34
33
|
const VARIATION_SELECTOR_16 = "\u{FE0F}";
|
|
35
|
-
|
|
34
|
+
const KEYCAP_BASE = /[0-9#*]/;
|
|
35
|
+
const COMBINING_KEYCAP = "\u{20E3}";
|
|
36
|
+
/**
|
|
37
|
+
* Every distinct emoji in a string, in TRUE order of first appearance.
|
|
38
|
+
*
|
|
39
|
+
* The keycap sequence ([0-9#*] + optional FE0F + 20E3) is detected in the
|
|
40
|
+
* scan loop at its real position. It used to be unshifted to the FRONT of the
|
|
41
|
+
* list regardless of where it sat, so the evidence named the wrong "first"
|
|
42
|
+
* emoji and anchored its excerpt on it (finding #9, 2026-09-26).
|
|
43
|
+
*/
|
|
36
44
|
function emojiIn(text) {
|
|
37
45
|
const found = [];
|
|
38
46
|
const chars = [...text];
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
}
|
|
47
|
+
const add = (glyph) => {
|
|
48
|
+
if (!found.includes(glyph))
|
|
49
|
+
found.push(glyph);
|
|
50
|
+
};
|
|
44
51
|
for (let i = 0; i < chars.length; i++) {
|
|
45
52
|
const ch = chars[i];
|
|
53
|
+
// Keycap first: it starts on an ordinary digit/#/* that the emoji tests
|
|
54
|
+
// below would not catch on its own.
|
|
55
|
+
if (KEYCAP_BASE.test(ch)) {
|
|
56
|
+
if (chars[i + 1] === COMBINING_KEYCAP) {
|
|
57
|
+
add(ch + chars[i + 1]);
|
|
58
|
+
i += 1;
|
|
59
|
+
continue;
|
|
60
|
+
}
|
|
61
|
+
if (chars[i + 1] === VARIATION_SELECTOR_16 && chars[i + 2] === COMBINING_KEYCAP) {
|
|
62
|
+
add(ch + chars[i + 1] + chars[i + 2]);
|
|
63
|
+
i += 2;
|
|
64
|
+
continue;
|
|
65
|
+
}
|
|
66
|
+
}
|
|
46
67
|
const isEmoji = DEFAULT_EMOJI.test(ch) ||
|
|
47
68
|
REGIONAL_INDICATOR.test(ch) ||
|
|
48
69
|
(PICTOGRAPHIC.test(ch) && chars[i + 1] === VARIATION_SELECTOR_16);
|
|
49
70
|
if (!isEmoji)
|
|
50
71
|
continue;
|
|
51
|
-
|
|
52
|
-
if (!found.includes(glyph))
|
|
53
|
-
found.push(glyph);
|
|
72
|
+
add(chars[i + 1] === VARIATION_SELECTOR_16 ? ch + chars[i + 1] : ch);
|
|
54
73
|
}
|
|
55
74
|
return found;
|
|
56
75
|
}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { violation } from "../types.js";
|
|
2
2
|
import { isProjectPath } from "./projectPaths.js";
|
|
3
|
-
import {
|
|
3
|
+
import { segments, leadingCommand, withoutCommitMessage } from "./shellCommand.js";
|
|
4
4
|
/**
|
|
5
5
|
* Third structured-check primitive: only counts real MUTATIONS of a
|
|
6
6
|
* protected file, never reads of it. Real false-positive this fixes
|
|
@@ -54,21 +54,35 @@ const CD_INTO_TEMP = /\bcd\s+["']?(?:\/private)?\/(?:tmp|var\/folders)\b|\bcd\s+
|
|
|
54
54
|
* path into a file is not mutating that path.
|
|
55
55
|
*/
|
|
56
56
|
function mutatesPathInBash(rawCommand, filePath) {
|
|
57
|
-
const command = withoutHeredocs(rawCommand);
|
|
58
57
|
const p = pathPattern(filePath);
|
|
59
|
-
const
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
58
|
+
const bare = escapeRegex(filePath.replace(/^\.\//, ""));
|
|
59
|
+
// A truncation/append redirect onto the path, with a trailing path boundary
|
|
60
|
+
// so `> CHANGELOG.md.new` does NOT count as touching `CHANGELOG.md` (#9).
|
|
61
|
+
const redirect = new RegExp(`>>?\\s*['"]?${bare}(?=$|[\\s'";)])`);
|
|
62
|
+
// sed only mutates with an in-place flag; matched as a real flag, not a bare
|
|
63
|
+
// "-i" substring that can sit inside a replacement like `s/api-id/…/` (#10).
|
|
64
|
+
const sedInPlace = /(?:^|\s)sed\b[^|;&]*\s(?:--in-place\b|-[A-Za-z]*i\b)/;
|
|
65
|
+
// Reason per SEGMENT: the mutating verb must be the segment's own leading
|
|
66
|
+
// command, so a path named inside a commit message, an echoed string, or
|
|
67
|
+
// another command's argument is not read as a mutation (#2). Heredoc bodies
|
|
68
|
+
// are already stripped by `segments`.
|
|
69
|
+
for (const raw of segments(rawCommand)) {
|
|
70
|
+
const seg = withoutCommitMessage(raw);
|
|
71
|
+
if (redirect.test(seg))
|
|
72
|
+
return true;
|
|
73
|
+
const exe = leadingCommand(seg);
|
|
74
|
+
if ((exe === "rm" || exe === "rmdir" || exe === "unlink") && new RegExp(`\\b(?:rm|rmdir|unlink)\\b[^|;&]*${p}`).test(seg))
|
|
75
|
+
return true;
|
|
76
|
+
if ((exe === "mv" || exe === "cp") && new RegExp(`\\b(?:mv|cp)\\b[^|;&]*${p}\\s*$`).test(seg))
|
|
77
|
+
return true;
|
|
78
|
+
if (exe === "tee" && new RegExp(`\\btee\\b[^|;&]*${p}`).test(seg))
|
|
79
|
+
return true;
|
|
80
|
+
if (exe === "truncate" && new RegExp(`\\btruncate\\b[^|;&]*${p}`).test(seg))
|
|
81
|
+
return true;
|
|
82
|
+
if (exe === "sed" && sedInPlace.test(seg) && new RegExp(p).test(seg))
|
|
83
|
+
return true;
|
|
84
|
+
}
|
|
85
|
+
return false;
|
|
72
86
|
}
|
|
73
87
|
function findMutation(events, filePath) {
|
|
74
88
|
const normalized = filePath.replace(/^\.\//, "");
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { violation } from "../types.js";
|
|
2
|
+
import { segments, leadingCommand } from "./shellCommand.js";
|
|
2
3
|
/**
|
|
3
4
|
* First real structured-check primitive: parses actual git command
|
|
4
5
|
* arguments instead of searching prose for a branch name as a substring.
|
|
@@ -28,18 +29,88 @@ import { violation } from "../types.js";
|
|
|
28
29
|
// becomes a real violation if a commit follows it — see
|
|
29
30
|
// findCheckoutCommitViolation.
|
|
30
31
|
const GIT_CHECKOUT_CREATE = /\bgit\s+(?:checkout\s+-[bB]|switch\s+-c)\s+([^\s-][^\s]*)/;
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
32
|
+
// A bare switch to an existing branch. NOTE: no `(?:--\s+)?` here — `git
|
|
33
|
+
// checkout -- <path>` restores a FILE (the `--` means "pathspec, not ref"),
|
|
34
|
+
// and reading that path as a branch caused a false FAIL when a file shared the
|
|
35
|
+
// protected branch's name (finding 2026-09-26).
|
|
36
|
+
const GIT_CHECKOUT_SWITCH_ONLY = /\bgit\s+(?:checkout|switch)\s+([^\s-][^\s]*)/;
|
|
34
37
|
const GIT_COMMIT = /\bgit\s+commit\b/;
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
38
|
+
// Locates the args after a `git … push`, allowing config flags between `git`
|
|
39
|
+
// and `push` (`git -c x=y push …`), the same shape as the other detectors.
|
|
40
|
+
const GIT_PUSH_ARGS = /\bgit\s+(?:\S+\s+){0,4}?push(?![\w-])\s*(.*)$/;
|
|
41
|
+
const GIT_BRANCH_ARGS = /\bgit\s+(?:\S+\s+){0,4}?branch(?![\w-])\s*(.*)$/;
|
|
42
|
+
/**
|
|
43
|
+
* The branch(es) a `git push` actually writes to. Parses the args instead of
|
|
44
|
+
* a single end-anchored regex, so it catches every real form the old GIT_PUSH
|
|
45
|
+
* missed: force-push, `-u`, a flag AFTER the branch, `--force-with-lease`, a
|
|
46
|
+
* remote-branch delete (`origin :main`, `origin --delete main`), and a
|
|
47
|
+
* `local:remote` refspec (the TARGET is the remote side).
|
|
48
|
+
*/
|
|
49
|
+
function pushTargets(command) {
|
|
50
|
+
const out = [];
|
|
51
|
+
for (const seg of segments(command)) {
|
|
52
|
+
if (leadingCommand(seg) !== "git")
|
|
53
|
+
continue;
|
|
54
|
+
const m = seg.match(GIT_PUSH_ARGS);
|
|
55
|
+
if (!m || m[1].trim().length === 0)
|
|
56
|
+
continue; // bare `git push`: branch unknown
|
|
57
|
+
const tokens = m[1].trim().split(/\s+/).filter(Boolean);
|
|
58
|
+
const positionals = tokens.filter((t) => !t.startsWith("-"));
|
|
59
|
+
// First positional is the remote; every later one is a refspec.
|
|
60
|
+
for (const ref of positionals.slice(1)) {
|
|
61
|
+
const spec = ref.replace(/^\+/, ""); // `+local:remote` force refspec
|
|
62
|
+
const colon = spec.indexOf(":");
|
|
63
|
+
const target = colon >= 0 ? spec.slice(colon + 1) : spec; // remote side
|
|
64
|
+
if (target.length > 0)
|
|
65
|
+
out.push(target);
|
|
66
|
+
}
|
|
41
67
|
}
|
|
42
|
-
return
|
|
68
|
+
return out;
|
|
69
|
+
}
|
|
70
|
+
/**
|
|
71
|
+
* The branch(es) a `git branch` command creates, deletes, or renames INTO.
|
|
72
|
+
* `-m/-M/--move <old> <new>` targets the NEW name (last positional — a rename
|
|
73
|
+
* into `main` overwrites `main`); `-d/-D/--delete` targets every named branch;
|
|
74
|
+
* a plain `git branch <name> [<start>]` targets only the created <name>.
|
|
75
|
+
*/
|
|
76
|
+
function branchTargets(command) {
|
|
77
|
+
const out = [];
|
|
78
|
+
for (const seg of segments(command)) {
|
|
79
|
+
if (leadingCommand(seg) !== "git")
|
|
80
|
+
continue;
|
|
81
|
+
const m = seg.match(GIT_BRANCH_ARGS);
|
|
82
|
+
if (!m || m[1].trim().length === 0)
|
|
83
|
+
continue;
|
|
84
|
+
const tokens = m[1].trim().split(/\s+/).filter(Boolean);
|
|
85
|
+
const flags = tokens.filter((t) => t.startsWith("-"));
|
|
86
|
+
const positionals = tokens.filter((t) => !t.startsWith("-"));
|
|
87
|
+
if (positionals.length === 0)
|
|
88
|
+
continue;
|
|
89
|
+
if (flags.some((f) => /^(?:--move|-m|-M)$/.test(f))) {
|
|
90
|
+
out.push(positionals[positionals.length - 1]);
|
|
91
|
+
}
|
|
92
|
+
else if (flags.some((f) => /^(?:--delete|-d|-D)$/.test(f))) {
|
|
93
|
+
out.push(...positionals);
|
|
94
|
+
}
|
|
95
|
+
else {
|
|
96
|
+
out.push(positionals[0]);
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
return out;
|
|
100
|
+
}
|
|
101
|
+
function extractGitBranchTargets(command) {
|
|
102
|
+
const hits = [];
|
|
103
|
+
const create = command.match(GIT_CHECKOUT_CREATE);
|
|
104
|
+
if (create)
|
|
105
|
+
hits.push({ branch: create[1], kind: "create" });
|
|
106
|
+
const switchOnly = command.match(GIT_CHECKOUT_SWITCH_ONLY);
|
|
107
|
+
if (switchOnly)
|
|
108
|
+
hits.push({ branch: switchOnly[1], kind: "switch" });
|
|
109
|
+
for (const b of branchTargets(command))
|
|
110
|
+
hits.push({ branch: b, kind: "create" });
|
|
111
|
+
for (const b of pushTargets(command))
|
|
112
|
+
hits.push({ branch: b, kind: "push" });
|
|
113
|
+
return hits;
|
|
43
114
|
}
|
|
44
115
|
function commandFromEvent(event) {
|
|
45
116
|
if (event.kind !== "tool_use" || event.toolName !== "Bash")
|
|
@@ -83,8 +154,8 @@ export function runGitBranchPolicyChecks(classifications, events) {
|
|
|
83
154
|
if (!command)
|
|
84
155
|
continue;
|
|
85
156
|
commands.push(command);
|
|
86
|
-
for (const
|
|
87
|
-
allTargets.push({ branch, command });
|
|
157
|
+
for (const hit of extractGitBranchTargets(command)) {
|
|
158
|
+
allTargets.push({ branch: hit.branch, command, kind: hit.kind });
|
|
88
159
|
}
|
|
89
160
|
}
|
|
90
161
|
const anyGitCommand = events.some((e) => e.kind === "tool_use" && /\bgit\s/.test(JSON.stringify(e.input ?? "")));
|
|
@@ -103,8 +174,9 @@ export function runGitBranchPolicyChecks(classifications, events) {
|
|
|
103
174
|
evidence: "no git command ran this session, so this rule never applied",
|
|
104
175
|
};
|
|
105
176
|
}
|
|
106
|
-
|
|
107
|
-
|
|
177
|
+
// Push, branch-create/delete/rename, and checkout -b are immediate hits.
|
|
178
|
+
// A bare `switch` is not — it only violates if a commit follows (below).
|
|
179
|
+
const pushOrCreateHit = allTargets.find((t) => t.branch === branchName && (t.kind === "push" || t.kind === "create"));
|
|
108
180
|
const commitViolationCommand = findCheckoutCommitViolation(commands, branchName);
|
|
109
181
|
const hit = pushOrCreateHit ?? (commitViolationCommand ? { branch: branchName, command: commitViolationCommand } : undefined);
|
|
110
182
|
if (polarity === "forbid") {
|
|
@@ -1,6 +1,12 @@
|
|
|
1
1
|
import { findTestRun } from "./testCommands.js";
|
|
2
2
|
import { isProjectPath } from "./projectPaths.js";
|
|
3
|
-
|
|
3
|
+
// Test-file conventions across ecosystems: JS `.test.`/`.spec.`, Go/Python
|
|
4
|
+
// `_test.`, `__tests__/` and `/tests/` dirs, RSpec `_spec.`/`/spec/`, and
|
|
5
|
+
// pytest's `test_*.py` (a `test_` prefix at a path boundary, so a prod file
|
|
6
|
+
// with "test" mid-name like `latest_data.py` is not swept in). Added
|
|
7
|
+
// pytest/RSpec 2026-09-26 — they were missing, so idiomatic Python/Ruby test
|
|
8
|
+
// files read as "no test touched".
|
|
9
|
+
const TEST_FILE_PATTERN = /(\.test\.|\.spec\.|_spec\.|__tests__\/|_test\.|(?:^|\/)test_|\/tests?\/|\/spec\/)/i;
|
|
4
10
|
// Real false-positive found 2026-08-30 on an actual session: editing a
|
|
5
11
|
// markdown documentation file flagged "no test file touched" four
|
|
6
12
|
// separate times — but a doc/config file was never going to have a test
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { segments, leadingCommand, withoutCommitMessage } from "./shellCommand.js";
|
|
2
2
|
import { canonicalise, matchesPattern } from "./deterministicChecks.js";
|
|
3
3
|
/**
|
|
4
4
|
* Commands that read. A banned literal appearing as an argument to one of
|
|
@@ -22,40 +22,26 @@ const READ_ONLY = new Set([
|
|
|
22
22
|
"awk", "jq", "yq", "cut", "tr", "column", "tee",
|
|
23
23
|
"which", "type", "file", "stat", "man", "help",
|
|
24
24
|
]);
|
|
25
|
-
/** Read-only git subcommands — `git log` cannot delete anything. */
|
|
26
|
-
const READ_ONLY_GIT = new Set(["log", "show", "diff", "status", "blame", "describe", "config", "remote", "branch", "tag", "ls-files", "rev-parse", "shortlog"]);
|
|
27
25
|
/**
|
|
28
|
-
*
|
|
26
|
+
* Genuinely read-only git subcommands — `git log` cannot delete anything.
|
|
29
27
|
*
|
|
30
|
-
*
|
|
31
|
-
*
|
|
32
|
-
*
|
|
33
|
-
*
|
|
28
|
+
* `branch`, `config`, `remote`, `tag` were removed 2026-09-26: each has a
|
|
29
|
+
* common MUTATING form (`git branch -D`, `git config user.email x`, `git
|
|
30
|
+
* remote set-url`, `git tag -f`) that the guard must be able to check, and a
|
|
31
|
+
* bare-flag literal (`-D`, `--force`) slipped past the old escape hatch.
|
|
34
32
|
*/
|
|
35
|
-
|
|
36
|
-
return withoutHeredocs(command)
|
|
37
|
-
.split(/\n|&&|\|\||[;|]/)
|
|
38
|
-
.map((s) => s.trim())
|
|
39
|
-
.filter((s) => s.length > 0);
|
|
40
|
-
}
|
|
41
|
-
/** The executable a segment invokes, with env assignments and `sudo` skipped. */
|
|
42
|
-
function leadingCommand(segment) {
|
|
43
|
-
const words = segment.split(/\s+/).filter(Boolean);
|
|
44
|
-
let i = 0;
|
|
45
|
-
while (i < words.length && (/^[A-Za-z_][A-Za-z0-9_]*=/.test(words[i]) || words[i] === "sudo" || words[i] === "command" || words[i] === "time"))
|
|
46
|
-
i++;
|
|
47
|
-
const exe = (words[i] ?? "").replace(/^.*\//, "");
|
|
48
|
-
return exe;
|
|
49
|
-
}
|
|
33
|
+
const READ_ONLY_GIT = new Set(["log", "show", "diff", "status", "blame", "describe", "ls-files", "rev-parse", "shortlog"]);
|
|
50
34
|
/**
|
|
51
|
-
* A
|
|
52
|
-
*
|
|
53
|
-
*
|
|
54
|
-
* backticked shell examples. A message that quotes a banned command in order
|
|
55
|
-
* to describe it must not be treated as running it.
|
|
35
|
+
* A normally-read-only command that, in this segment, actually EXECUTES or
|
|
36
|
+
* DELETES: `find … -exec/-execdir/-delete`, `awk 'BEGIN{system("…")}'`.
|
|
37
|
+
* Without this, a destructive command wrapped in one was read as harmless.
|
|
56
38
|
*/
|
|
57
|
-
function
|
|
58
|
-
|
|
39
|
+
function readOnlyCommandActuallyRuns(exe, segment) {
|
|
40
|
+
if (exe === "find" && /\s-(?:exec(?:dir)?|delete)\b/.test(segment))
|
|
41
|
+
return true;
|
|
42
|
+
if (exe === "awk" && /\bsystem\s*\(/.test(segment))
|
|
43
|
+
return true;
|
|
44
|
+
return false;
|
|
59
45
|
}
|
|
60
46
|
/**
|
|
61
47
|
* Does this segment actually DO the forbidden thing?
|
|
@@ -64,9 +50,11 @@ function withoutCommitMessage(segment) {
|
|
|
64
50
|
*/
|
|
65
51
|
function segmentRunsLiteral(segment, literal) {
|
|
66
52
|
const exe = leadingCommand(segment);
|
|
67
|
-
if (READ_ONLY.has(exe))
|
|
53
|
+
if (READ_ONLY.has(exe) && !readOnlyCommandActuallyRuns(exe, segment))
|
|
68
54
|
return false;
|
|
69
|
-
|
|
55
|
+
// Plain `sed` prints to stdout; only an in-place edit mutates. Both the
|
|
56
|
+
// short `-i`/`-i.bak` and the long `--in-place` spellings count.
|
|
57
|
+
if (exe === "sed" && !/\s--in-place\b|\s-[A-Za-z]*i\b/.test(segment))
|
|
70
58
|
return false;
|
|
71
59
|
if (exe === "git") {
|
|
72
60
|
const sub = segment.split(/\s+/).filter(Boolean)[1] ?? "";
|
|
@@ -22,3 +22,25 @@
|
|
|
22
22
|
* heredoc that set the fixture up.
|
|
23
23
|
*/
|
|
24
24
|
export declare function withoutHeredocs(command: string): string;
|
|
25
|
+
/**
|
|
26
|
+
* Splits a shell command into the pieces that run separately.
|
|
27
|
+
*
|
|
28
|
+
* Crude by design — not a shell parser. Heredoc bodies are stripped first, so
|
|
29
|
+
* a command that only WRITES another command is not split into it. Shared by
|
|
30
|
+
* every checker that needs to reason about what a compound command actually
|
|
31
|
+
* runs (proposedAction, attribution).
|
|
32
|
+
*/
|
|
33
|
+
export declare function segments(command: string): string[];
|
|
34
|
+
/** The executable a segment invokes, with env assignments and `sudo` skipped. */
|
|
35
|
+
export declare function leadingCommand(segment: string): string;
|
|
36
|
+
/**
|
|
37
|
+
* Blanks out a git/gh commit or PR MESSAGE, leaving the command around it.
|
|
38
|
+
*
|
|
39
|
+
* A `-m "…"` / `--message "…"` value is text the author wrote, not a command
|
|
40
|
+
* being run, and not a path being touched — a message that quotes `rm …` or
|
|
41
|
+
* names a protected file must not be read as doing either. Shared so every
|
|
42
|
+
* checker that scans a Bash command string strips it the same way (this was
|
|
43
|
+
* present only in proposedAction, so fileLifecycle and testCommands each
|
|
44
|
+
* false-matched on commit-message text — findings 2026-09-26).
|
|
45
|
+
*/
|
|
46
|
+
export declare function withoutCommitMessage(command: string): string;
|