claude-dev-env 8.32.1 → 8.33.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/_shared/pr-loop/scripts/test_fix_pr_test_proof.py +51 -0
- package/hooks/advisory/pr_done_reminder.py +16 -75
- package/hooks/advisory/test_pr_done_reminder.py +19 -0
- package/hooks/blocking/test_verify_before_acting.py +655 -0
- package/hooks/blocking/verify_before_acting.py +323 -0
- package/hooks/hooks.json +10 -0
- package/hooks/hooks_constants/piped_pytest_blocker_constants.py +78 -6
- package/hooks/hooks_constants/pr_done_reminder_constants.py +0 -8
- package/hooks/hooks_constants/pytest_invocation.py +60 -22
- package/hooks/hooks_constants/shell_command_mutation.py +202 -0
- package/hooks/hooks_constants/shell_command_segments.py +39 -0
- package/hooks/hooks_constants/shell_command_wrappers.py +68 -0
- package/hooks/hooks_constants/test_pr_done_reminder_constants.py +0 -11
- package/hooks/hooks_constants/test_pytest_invocation.py +6 -0
- package/hooks/hooks_constants/test_shell_command_mutation.py +37 -0
- package/hooks/hooks_constants/test_shell_command_segments.py +29 -0
- package/hooks/hooks_constants/test_shell_command_wrappers.py +61 -0
- package/hooks/hooks_constants/verify_before_acting_constants.py +170 -0
- package/package.json +1 -1
- package/scripts/policy_lint/adapter_configuration.py +4 -0
- package/scripts/policy_lint/config/constants.py +1 -0
- package/scripts/tests/test_adapter_configuration.py +8 -0
|
@@ -0,0 +1,655 @@
|
|
|
1
|
+
import io
|
|
2
|
+
import json
|
|
3
|
+
import re
|
|
4
|
+
import threading
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
import pytest
|
|
8
|
+
|
|
9
|
+
import verify_before_acting
|
|
10
|
+
|
|
11
|
+
ACTING_MESSAGE_ID = "msg_acting"
|
|
12
|
+
EARLIER_MESSAGE_ID = "msg_earlier"
|
|
13
|
+
TOOL_USE_ID = "toolu_target"
|
|
14
|
+
HEDGED_SENTENCE = "The config probably lives in settings.json."
|
|
15
|
+
CLEAN_SENTENCE = "The config lives in settings.json, as the read above showed."
|
|
16
|
+
LOG_RELATIVE_PATH = Path(".claude") / "logs" / "verify-before-acting.jsonl"
|
|
17
|
+
WRITE_INPUT = {"file_path": "settings.json", "content": "{}"}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def thinking_record(message_id: str, thinking_text: str) -> dict[str, object]:
|
|
21
|
+
return {
|
|
22
|
+
"type": "assistant",
|
|
23
|
+
"message": {
|
|
24
|
+
"id": message_id,
|
|
25
|
+
"role": "assistant",
|
|
26
|
+
"content": [{"type": "thinking", "thinking": thinking_text, "signature": "sig"}],
|
|
27
|
+
},
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def tool_use_record(
|
|
32
|
+
message_id: str, tool_use_id: str, tool_name: str, tool_input: dict[str, object]
|
|
33
|
+
) -> dict[str, object]:
|
|
34
|
+
return {
|
|
35
|
+
"type": "assistant",
|
|
36
|
+
"message": {
|
|
37
|
+
"id": message_id,
|
|
38
|
+
"role": "assistant",
|
|
39
|
+
"content": [
|
|
40
|
+
{"type": "tool_use", "id": tool_use_id, "name": tool_name, "input": tool_input}
|
|
41
|
+
],
|
|
42
|
+
},
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def tool_result_record(tool_use_id: str) -> dict[str, object]:
|
|
47
|
+
return {
|
|
48
|
+
"type": "user",
|
|
49
|
+
"message": {
|
|
50
|
+
"role": "user",
|
|
51
|
+
"content": [{"type": "tool_result", "tool_use_id": tool_use_id, "content": "ok"}],
|
|
52
|
+
},
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def write_transcript(directory: Path, all_lines: list[object]) -> Path:
|
|
57
|
+
transcript_path = directory / "transcript.jsonl"
|
|
58
|
+
all_texts = [
|
|
59
|
+
each_line if isinstance(each_line, str) else json.dumps(each_line)
|
|
60
|
+
for each_line in all_lines
|
|
61
|
+
]
|
|
62
|
+
transcript_path.write_text("\n".join(all_texts) + "\n", encoding="utf-8")
|
|
63
|
+
return transcript_path
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def acting_transcript(
|
|
67
|
+
directory: Path, thinking_text: str, tool_name: str, tool_input: dict[str, object]
|
|
68
|
+
) -> Path:
|
|
69
|
+
return write_transcript(
|
|
70
|
+
directory,
|
|
71
|
+
[
|
|
72
|
+
thinking_record(ACTING_MESSAGE_ID, thinking_text),
|
|
73
|
+
tool_use_record(ACTING_MESSAGE_ID, TOOL_USE_ID, tool_name, tool_input),
|
|
74
|
+
],
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def run_hook(
|
|
79
|
+
monkeypatch: pytest.MonkeyPatch,
|
|
80
|
+
capsys: pytest.CaptureFixture[str],
|
|
81
|
+
tool_name: str,
|
|
82
|
+
tool_input: dict[str, object],
|
|
83
|
+
transcript_path: Path,
|
|
84
|
+
) -> tuple[int, str]:
|
|
85
|
+
hook_input = {
|
|
86
|
+
"hook_event_name": "PostToolUse",
|
|
87
|
+
"tool_name": tool_name,
|
|
88
|
+
"tool_input": tool_input,
|
|
89
|
+
"tool_use_id": TOOL_USE_ID,
|
|
90
|
+
"transcript_path": str(transcript_path),
|
|
91
|
+
}
|
|
92
|
+
monkeypatch.setattr(
|
|
93
|
+
"sys.stdin", io.TextIOWrapper(io.BytesIO(json.dumps(hook_input).encode("utf-8")))
|
|
94
|
+
)
|
|
95
|
+
exit_code = verify_before_acting.main()
|
|
96
|
+
return exit_code, capsys.readouterr().out
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def logged_outcomes(home_directory: Path) -> list[str]:
|
|
100
|
+
log_path = home_directory / LOG_RELATIVE_PATH
|
|
101
|
+
if not log_path.exists():
|
|
102
|
+
return []
|
|
103
|
+
all_records = [
|
|
104
|
+
json.loads(each_line)
|
|
105
|
+
for each_line in log_path.read_text(encoding="utf-8").splitlines()
|
|
106
|
+
if each_line
|
|
107
|
+
]
|
|
108
|
+
return [each_record["outcome"] for each_record in all_records]
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def expected_block(tool_name: str, quoted_sentence: str) -> dict[str, object]:
|
|
112
|
+
return {
|
|
113
|
+
"decision": "block",
|
|
114
|
+
"reason": (
|
|
115
|
+
f'The reasoning behind this {tool_name} call hedges: "{quoted_sentence}" '
|
|
116
|
+
"Check that claim now with a read-only tool, then continue. "
|
|
117
|
+
"If the check contradicts it, undo this change first."
|
|
118
|
+
),
|
|
119
|
+
"hookSpecificOutput": {"hookEventName": "PostToolUse"},
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
@pytest.fixture(autouse=True)
|
|
124
|
+
def isolated_home(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> None:
|
|
125
|
+
monkeypatch.setenv("HOME", str(tmp_path))
|
|
126
|
+
monkeypatch.setenv("USERPROFILE", str(tmp_path))
|
|
127
|
+
monkeypatch.setattr(verify_before_acting, "TRANSCRIPT_POLL_LIMIT_SECONDS", 0.3)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def test_should_block_a_hedged_write_whose_record_lands_after_the_hook_starts(
|
|
131
|
+
monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
|
|
132
|
+
) -> None:
|
|
133
|
+
transcript_path = write_transcript(tmp_path, [thinking_record(EARLIER_MESSAGE_ID, "Start.")])
|
|
134
|
+
late_lines = "".join(
|
|
135
|
+
json.dumps(each_record) + "\n"
|
|
136
|
+
for each_record in (
|
|
137
|
+
thinking_record(ACTING_MESSAGE_ID, HEDGED_SENTENCE),
|
|
138
|
+
tool_use_record(ACTING_MESSAGE_ID, TOOL_USE_ID, "Write", WRITE_INPUT),
|
|
139
|
+
)
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
def append_late_lines() -> None:
|
|
143
|
+
with transcript_path.open("a", encoding="utf-8") as transcript_file:
|
|
144
|
+
transcript_file.write(late_lines)
|
|
145
|
+
|
|
146
|
+
writer = threading.Timer(0.15, append_late_lines)
|
|
147
|
+
writer.start()
|
|
148
|
+
exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
|
|
149
|
+
writer.join()
|
|
150
|
+
|
|
151
|
+
assert exit_code == 0
|
|
152
|
+
assert json.loads(stdout_text) == expected_block("Write", HEDGED_SENTENCE)
|
|
153
|
+
assert logged_outcomes(tmp_path) == ["blocked"]
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def test_should_allow_and_log_unseen_when_the_record_never_lands(
|
|
157
|
+
monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
|
|
158
|
+
) -> None:
|
|
159
|
+
transcript_path = write_transcript(tmp_path, [thinking_record(EARLIER_MESSAGE_ID, "Start.")])
|
|
160
|
+
exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
|
|
161
|
+
assert (exit_code, stdout_text) == (0, "")
|
|
162
|
+
assert logged_outcomes(tmp_path) == ["reasoning_unseen"]
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def test_should_block_a_write_after_hedged_reasoning(
|
|
166
|
+
monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
|
|
167
|
+
) -> None:
|
|
168
|
+
transcript_path = acting_transcript(
|
|
169
|
+
tmp_path, f"I need the config.\n{HEDGED_SENTENCE} Writing it now.", "Write", WRITE_INPUT
|
|
170
|
+
)
|
|
171
|
+
exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
|
|
172
|
+
assert exit_code == 0
|
|
173
|
+
assert json.loads(stdout_text) == expected_block("Write", HEDGED_SENTENCE)
|
|
174
|
+
assert logged_outcomes(tmp_path) == ["blocked"]
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def test_should_log_the_quoted_hedge_sentence_on_a_block(
|
|
178
|
+
monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
|
|
179
|
+
) -> None:
|
|
180
|
+
transcript_path = acting_transcript(tmp_path, HEDGED_SENTENCE, "Write", WRITE_INPUT)
|
|
181
|
+
run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
|
|
182
|
+
log_line = (tmp_path / LOG_RELATIVE_PATH).read_text(encoding="utf-8").splitlines()[0]
|
|
183
|
+
log_record = json.loads(log_line)
|
|
184
|
+
assert log_record["tool_name"] == "Write"
|
|
185
|
+
assert log_record["tool_use_id"] == TOOL_USE_ID
|
|
186
|
+
assert log_record["hedge_sentence"] == HEDGED_SENTENCE
|
|
187
|
+
assert log_record["timestamp"]
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def test_should_allow_a_write_after_clean_reasoning(
|
|
191
|
+
monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
|
|
192
|
+
) -> None:
|
|
193
|
+
transcript_path = acting_transcript(tmp_path, CLEAN_SENTENCE, "Write", WRITE_INPUT)
|
|
194
|
+
exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
|
|
195
|
+
assert (exit_code, stdout_text) == (0, "")
|
|
196
|
+
assert logged_outcomes(tmp_path) == ["allowed_clean"]
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
@pytest.mark.parametrize(
|
|
200
|
+
("tool_name", "command"),
|
|
201
|
+
[
|
|
202
|
+
("Bash", "gh pr view 12 --json state"),
|
|
203
|
+
("PowerShell", "Get-Content settings.json"),
|
|
204
|
+
("PowerShell", "Test-Path settings.json"),
|
|
205
|
+
("Bash", "gh api repos/o/r/pulls/12"),
|
|
206
|
+
("Bash", "git status 2>&1"),
|
|
207
|
+
("Bash", "git log --oneline > /dev/null"),
|
|
208
|
+
("Bash", "cat notes.txt"),
|
|
209
|
+
("Bash", "cat touch.txt"),
|
|
210
|
+
("Bash", "ls -la"),
|
|
211
|
+
("Bash", "ls mkdir_notes"),
|
|
212
|
+
("Bash", "grep -rn rm src"),
|
|
213
|
+
("Bash", 'grep -n "cp " notes.md'),
|
|
214
|
+
("Bash", "git status"),
|
|
215
|
+
("Bash", "git diff"),
|
|
216
|
+
("Bash", "git diff -- cp.py"),
|
|
217
|
+
("Bash", "gh pr view 12"),
|
|
218
|
+
("Bash", "sed -n 1,5p notes.txt"),
|
|
219
|
+
("Bash", "sed --silent 1p notes.txt"),
|
|
220
|
+
("Bash", "git log --oneline -- rm.py"),
|
|
221
|
+
("Bash", "python pull_request.py --help"),
|
|
222
|
+
("Bash", "awk '$1 > 5' notes.txt"),
|
|
223
|
+
("Bash", "jq '.count > 3' data.json"),
|
|
224
|
+
("Bash", "gh pr list --json number --jq 'map(select(.number > 5))'"),
|
|
225
|
+
("Bash", 'python -c "print(1 > 0)"'),
|
|
226
|
+
("Bash", "grep -c '>' notes.txt"),
|
|
227
|
+
("Bash", "git log --format='%h > %s'"),
|
|
228
|
+
("Bash", "gh api -X GET search/issues -f q=repo:o/r"),
|
|
229
|
+
("Bash", "gh api --method GET repos/o/r/pulls -f state=open"),
|
|
230
|
+
("Bash", "git --no-pager log"),
|
|
231
|
+
("Bash", "git --git-dir=/repo/.git log --oneline"),
|
|
232
|
+
("Bash", "git -C /repo stash list"),
|
|
233
|
+
("Bash", "git -c core.pager=cat diff"),
|
|
234
|
+
("Bash", 'pwsh -NoProfile -Command "git status"'),
|
|
235
|
+
("PowerShell", "Get-Content x.txt > $null"),
|
|
236
|
+
("Bash", "rg 'gh pr merge' docs"),
|
|
237
|
+
("Bash", 'grep -rn "gh issue comment" docs'),
|
|
238
|
+
("Bash", "rg 'gh api repos/o/r/issues -f title=x' docs"),
|
|
239
|
+
("Bash", "rg 'gh workflow run' .github"),
|
|
240
|
+
("Bash", "rg 'Remove-Item' docs"),
|
|
241
|
+
("PowerShell", "Select-String -Pattern 'Set-Content' -Path notes.md"),
|
|
242
|
+
("Bash", 'grep "a|cp b" notes.md'),
|
|
243
|
+
("Bash", "grep 'x; rm y' notes.md"),
|
|
244
|
+
("Bash", "rg 'sed -i' docs"),
|
|
245
|
+
("Bash", "rg 'pull_request.py create' docs"),
|
|
246
|
+
("Bash", "sudo cat notes.txt"),
|
|
247
|
+
("Bash", 'bash -c "git status"'),
|
|
248
|
+
("Bash", "sh -c ls"),
|
|
249
|
+
("Bash", "bash -lc 'git log --oneline'"),
|
|
250
|
+
("Bash", "env VAR=x git status"),
|
|
251
|
+
("Bash", "echo HEAD | xargs git show"),
|
|
252
|
+
("Bash", "pwsh -Command git status"),
|
|
253
|
+
("Bash", "pwsh -Command Get-Content notes.md"),
|
|
254
|
+
("Bash", "env -u HOME git status"),
|
|
255
|
+
("Bash", "timeout --signal KILL 5 git status"),
|
|
256
|
+
("Bash", "printf x | xargs -I {} cat {}"),
|
|
257
|
+
("Bash", "printf x | xargs -n 1 git show"),
|
|
258
|
+
("PowerShell", "Get-Command Remove-Item"),
|
|
259
|
+
("PowerShell", "Get-Help Set-Content -Full"),
|
|
260
|
+
("Bash", "echo hi >&-"),
|
|
261
|
+
("Bash", "echo hi 1>&2"),
|
|
262
|
+
("Bash", "git.exe status"),
|
|
263
|
+
("Bash", "gh pr view 8 --json reviews"),
|
|
264
|
+
],
|
|
265
|
+
)
|
|
266
|
+
def test_should_pass_a_read_only_command_untouched(
|
|
267
|
+
monkeypatch: pytest.MonkeyPatch,
|
|
268
|
+
capsys: pytest.CaptureFixture[str],
|
|
269
|
+
tmp_path: Path,
|
|
270
|
+
tool_name: str,
|
|
271
|
+
command: str,
|
|
272
|
+
) -> None:
|
|
273
|
+
tool_input = {"command": command}
|
|
274
|
+
transcript_path = acting_transcript(tmp_path, HEDGED_SENTENCE, tool_name, tool_input)
|
|
275
|
+
exit_code, stdout_text = run_hook(monkeypatch, capsys, tool_name, tool_input, transcript_path)
|
|
276
|
+
assert (exit_code, stdout_text) == (0, "")
|
|
277
|
+
assert not (tmp_path / LOG_RELATIVE_PATH).exists()
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
@pytest.mark.parametrize(
|
|
281
|
+
("tool_name", "command"),
|
|
282
|
+
[
|
|
283
|
+
("Bash", "git push origin feat/x"),
|
|
284
|
+
("Bash", 'git -C "C:/repo dir" commit -F message.txt'),
|
|
285
|
+
("Bash", "gh pr merge 12 --squash"),
|
|
286
|
+
("Bash", "gh issue comment 4 --body-file body.md"),
|
|
287
|
+
("Bash", "gh api repos/o/r/issues/4/comments -X POST"),
|
|
288
|
+
("Bash", "gh api repos/o/r/pulls/12 --method PATCH"),
|
|
289
|
+
("Bash", "gh api repos/o/r/issues -f title=x"),
|
|
290
|
+
("Bash", "gh run rerun 99"),
|
|
291
|
+
("Bash", "gh workflow run ci.yml"),
|
|
292
|
+
("PowerShell", "Remove-Item -Force C:/scratch/x.txt"),
|
|
293
|
+
("PowerShell", "set-content x.txt 'y'"),
|
|
294
|
+
("Bash", "echo hi > notes.txt"),
|
|
295
|
+
("Bash", "rm notes.txt"),
|
|
296
|
+
("Bash", "rm -rf build"),
|
|
297
|
+
("Bash", "rmdir out"),
|
|
298
|
+
("Bash", "unlink link.txt"),
|
|
299
|
+
("Bash", "touch notes.txt"),
|
|
300
|
+
("Bash", "cp source.txt notes.txt"),
|
|
301
|
+
("Bash", "mv old.txt new.txt"),
|
|
302
|
+
("Bash", "mkdir -p out"),
|
|
303
|
+
("Bash", "echo hi | tee notes.txt"),
|
|
304
|
+
("Bash", "sed -i 's/a/b/' notes.txt"),
|
|
305
|
+
("Bash", "sed -i.bak 's/a/b/' notes.txt"),
|
|
306
|
+
("Bash", "sed -e 's/a/b/' -i notes.txt"),
|
|
307
|
+
("Bash", "sed --in-place 's/a/b/' notes.txt"),
|
|
308
|
+
("Bash", "ln notes.txt link.txt"),
|
|
309
|
+
("Bash", "ln -s target link"),
|
|
310
|
+
("Bash", "ln -sf target link"),
|
|
311
|
+
("Bash", "git status && rm notes.txt"),
|
|
312
|
+
("Bash", "sudo rm notes.txt"),
|
|
313
|
+
("Bash", "find . -name '*.tmp' | xargs rm"),
|
|
314
|
+
("Bash", "/bin/rm notes.txt"),
|
|
315
|
+
("Bash", "python ~/.agents/skills/pull-request/scripts/pull_request.py create --repo o/r"),
|
|
316
|
+
("Bash", "python pull_request.py edit --repo o/r --number 12 --body-file body.md"),
|
|
317
|
+
("Bash", "python pull_request.py comment --repo o/r --number 12 --body-file body.md"),
|
|
318
|
+
("Bash", "python pull_request.py review --repo o/r --number 12 --event approve"),
|
|
319
|
+
("Bash", 'pwsh -NoProfile -Command "git push origin HEAD"'),
|
|
320
|
+
("Bash", "cd repo && git checkout -b feat/x"),
|
|
321
|
+
("Bash", "echo hi>notes.txt"),
|
|
322
|
+
("Bash", "echo hi >> notes.txt"),
|
|
323
|
+
("Bash", "make 2> errors.log"),
|
|
324
|
+
("Bash", "make &> out.log"),
|
|
325
|
+
("PowerShell", "rm C:/scratch/x.txt"),
|
|
326
|
+
("PowerShell", "mkdir out"),
|
|
327
|
+
("Bash", "pwsh -Command git push origin HEAD"),
|
|
328
|
+
("Bash", "pwsh -Command git push origin HEAD; echo done"),
|
|
329
|
+
("Bash", 'pwsh -NoProfile -Command "Get-ChildItem *.tmp | Remove-Item"'),
|
|
330
|
+
("PowerShell", "Get-ChildItem *.tmp | ForEach-Object { Remove-Item $_ }"),
|
|
331
|
+
("Bash", "bash -c 'gh pr merge 12 --squash'"),
|
|
332
|
+
("Bash", "sudo gh api repos/o/r/issues -X POST"),
|
|
333
|
+
("Bash", "gh api repos/o/r/issues -XPOST"),
|
|
334
|
+
("Bash", "gh api repos/o/r/pulls/12 --method=PATCH"),
|
|
335
|
+
("Bash", "printf x | xargs -r rm"),
|
|
336
|
+
("Bash", "printf x | xargs -I {} rm {}"),
|
|
337
|
+
("Bash", "printf x | xargs -i rm {}"),
|
|
338
|
+
("Bash", "printf x | xargs -n 1 rm"),
|
|
339
|
+
("Bash", "find . -name '*.tmp' -print0 | xargs -0 rm"),
|
|
340
|
+
("Bash", "gh api repos/o/r/issues --input payload.json"),
|
|
341
|
+
("Bash", "gh api repos/o/r/issues --input=payload.json"),
|
|
342
|
+
("Bash", "git.exe push origin HEAD"),
|
|
343
|
+
("Bash", "gh.exe pr merge 12 --squash"),
|
|
344
|
+
("PowerShell", "& 'C:/Program Files/Git/cmd/git.exe' commit -F message.txt"),
|
|
345
|
+
("Bash", "echo hi > 1"),
|
|
346
|
+
("Bash", "echo hi > -"),
|
|
347
|
+
("Bash", "echo hi >| notes.txt"),
|
|
348
|
+
("Bash", "echo hi>|notes.txt"),
|
|
349
|
+
("Bash", "gh pr review 8 --approve"),
|
|
350
|
+
("PowerShell", "Get-ChildItem *.tmp | % {Remove-Item $_}"),
|
|
351
|
+
],
|
|
352
|
+
)
|
|
353
|
+
def test_should_block_a_mutating_command_after_hedged_reasoning(
|
|
354
|
+
monkeypatch: pytest.MonkeyPatch,
|
|
355
|
+
capsys: pytest.CaptureFixture[str],
|
|
356
|
+
tmp_path: Path,
|
|
357
|
+
tool_name: str,
|
|
358
|
+
command: str,
|
|
359
|
+
) -> None:
|
|
360
|
+
tool_input = {"command": command}
|
|
361
|
+
transcript_path = acting_transcript(tmp_path, HEDGED_SENTENCE, tool_name, tool_input)
|
|
362
|
+
exit_code, stdout_text = run_hook(monkeypatch, capsys, tool_name, tool_input, transcript_path)
|
|
363
|
+
assert exit_code == 0
|
|
364
|
+
assert json.loads(stdout_text) == expected_block(tool_name, HEDGED_SENTENCE)
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
@pytest.mark.parametrize(
|
|
368
|
+
"global_options",
|
|
369
|
+
[
|
|
370
|
+
"--git-dir=/repo/.git",
|
|
371
|
+
"--git-dir /repo/.git",
|
|
372
|
+
"--work-tree=/repo",
|
|
373
|
+
"--work-tree /repo",
|
|
374
|
+
"-C /repo",
|
|
375
|
+
"-c user.name=x",
|
|
376
|
+
"--no-pager",
|
|
377
|
+
"--namespace=n",
|
|
378
|
+
"--namespace n",
|
|
379
|
+
"-c user.name=x -C /repo",
|
|
380
|
+
],
|
|
381
|
+
)
|
|
382
|
+
@pytest.mark.parametrize("subcommand", ["push origin HEAD", "commit -F message.txt"])
|
|
383
|
+
def test_should_block_a_git_change_behind_global_options(
|
|
384
|
+
monkeypatch: pytest.MonkeyPatch,
|
|
385
|
+
capsys: pytest.CaptureFixture[str],
|
|
386
|
+
tmp_path: Path,
|
|
387
|
+
global_options: str,
|
|
388
|
+
subcommand: str,
|
|
389
|
+
) -> None:
|
|
390
|
+
tool_input = {"command": f"git {global_options} {subcommand}"}
|
|
391
|
+
transcript_path = acting_transcript(tmp_path, HEDGED_SENTENCE, "Bash", tool_input)
|
|
392
|
+
exit_code, stdout_text = run_hook(monkeypatch, capsys, "Bash", tool_input, transcript_path)
|
|
393
|
+
assert exit_code == 0
|
|
394
|
+
assert json.loads(stdout_text) == expected_block("Bash", HEDGED_SENTENCE)
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
@pytest.mark.parametrize(
|
|
398
|
+
"command_template",
|
|
399
|
+
[
|
|
400
|
+
"sudo git {subcommand}",
|
|
401
|
+
"sudo -u root git {subcommand}",
|
|
402
|
+
'bash -c "git {subcommand}"',
|
|
403
|
+
"sh -c 'git {subcommand}'",
|
|
404
|
+
"bash -lc 'git {subcommand}'",
|
|
405
|
+
"env VAR=x git {subcommand}",
|
|
406
|
+
"echo origin | xargs git {subcommand}",
|
|
407
|
+
"pwsh -Command git {subcommand}",
|
|
408
|
+
"sudo bash -c 'git {subcommand}'",
|
|
409
|
+
"env -u HOME git {subcommand}",
|
|
410
|
+
"env -C /repo git {subcommand}",
|
|
411
|
+
"timeout --signal KILL 5 git {subcommand}",
|
|
412
|
+
"timeout -k 2 5 git {subcommand}",
|
|
413
|
+
"timeout 5 git {subcommand}",
|
|
414
|
+
"echo origin | xargs -n 1 git {subcommand}",
|
|
415
|
+
"echo origin | xargs -r git {subcommand}",
|
|
416
|
+
"echo origin | xargs -I {{}} git {subcommand}",
|
|
417
|
+
"echo origin | xargs -0 -P 4 git {subcommand}",
|
|
418
|
+
],
|
|
419
|
+
)
|
|
420
|
+
@pytest.mark.parametrize("subcommand", ["push origin HEAD", "commit -F message.txt"])
|
|
421
|
+
def test_should_block_a_git_change_behind_a_wrapper(
|
|
422
|
+
monkeypatch: pytest.MonkeyPatch,
|
|
423
|
+
capsys: pytest.CaptureFixture[str],
|
|
424
|
+
tmp_path: Path,
|
|
425
|
+
command_template: str,
|
|
426
|
+
subcommand: str,
|
|
427
|
+
) -> None:
|
|
428
|
+
tool_input = {"command": command_template.format(subcommand=subcommand)}
|
|
429
|
+
transcript_path = acting_transcript(tmp_path, HEDGED_SENTENCE, "Bash", tool_input)
|
|
430
|
+
exit_code, stdout_text = run_hook(monkeypatch, capsys, "Bash", tool_input, transcript_path)
|
|
431
|
+
assert exit_code == 0
|
|
432
|
+
assert json.loads(stdout_text) == expected_block("Bash", HEDGED_SENTENCE)
|
|
433
|
+
|
|
434
|
+
|
|
435
|
+
@pytest.mark.parametrize(
|
|
436
|
+
"tool_name",
|
|
437
|
+
[
|
|
438
|
+
"Edit",
|
|
439
|
+
"MultiEdit",
|
|
440
|
+
"NotebookEdit",
|
|
441
|
+
"Agent",
|
|
442
|
+
"Task",
|
|
443
|
+
"apply_patch",
|
|
444
|
+
"mcp__gmail__send_message",
|
|
445
|
+
"mcp__github__issue_write",
|
|
446
|
+
"mcp__github__sub_issue_write",
|
|
447
|
+
"mcp__github__pull_request_review_write",
|
|
448
|
+
"mcp__github__add_issue_comment",
|
|
449
|
+
"mcp__atlassian__createJiraIssue",
|
|
450
|
+
"mcp__trello__trelloWriteCard",
|
|
451
|
+
"mcp__github__request_copilot_review",
|
|
452
|
+
],
|
|
453
|
+
)
|
|
454
|
+
def test_should_block_each_always_mutating_tool_after_hedged_reasoning(
|
|
455
|
+
monkeypatch: pytest.MonkeyPatch,
|
|
456
|
+
capsys: pytest.CaptureFixture[str],
|
|
457
|
+
tmp_path: Path,
|
|
458
|
+
tool_name: str,
|
|
459
|
+
) -> None:
|
|
460
|
+
transcript_path = acting_transcript(tmp_path, HEDGED_SENTENCE, tool_name, {})
|
|
461
|
+
exit_code, stdout_text = run_hook(monkeypatch, capsys, tool_name, {}, transcript_path)
|
|
462
|
+
assert exit_code == 0
|
|
463
|
+
assert json.loads(stdout_text)["decision"] == "block"
|
|
464
|
+
|
|
465
|
+
|
|
466
|
+
def test_should_register_a_matcher_covering_every_mutating_tool_name() -> None:
|
|
467
|
+
hooks_configuration = json.loads(
|
|
468
|
+
(Path(__file__).resolve().parent.parent / "hooks.json").read_text(encoding="utf-8")
|
|
469
|
+
)
|
|
470
|
+
all_matchers = [
|
|
471
|
+
each_entry["matcher"]
|
|
472
|
+
for each_entry in hooks_configuration["hooks"]["PostToolUse"]
|
|
473
|
+
if any(
|
|
474
|
+
"verify_before_acting.py" in each_hook["command"] for each_hook in each_entry["hooks"]
|
|
475
|
+
)
|
|
476
|
+
]
|
|
477
|
+
assert len(all_matchers) == 1
|
|
478
|
+
matcher_pattern = re.compile(all_matchers[0])
|
|
479
|
+
all_expected_names = (
|
|
480
|
+
verify_before_acting.ALL_ALWAYS_MUTATING_TOOL_NAMES
|
|
481
|
+
| verify_before_acting.ALL_SHELL_TOOL_NAMES
|
|
482
|
+
)
|
|
483
|
+
assert {
|
|
484
|
+
each_name for each_name in all_expected_names if not matcher_pattern.fullmatch(each_name)
|
|
485
|
+
} == set()
|
|
486
|
+
|
|
487
|
+
|
|
488
|
+
@pytest.mark.parametrize(
|
|
489
|
+
"tool_name",
|
|
490
|
+
[
|
|
491
|
+
"TodoWrite",
|
|
492
|
+
"TaskUpdate",
|
|
493
|
+
"Read",
|
|
494
|
+
"mcp__gmail__get_thread",
|
|
495
|
+
"mcp__github__issue_read",
|
|
496
|
+
"mcp__github__list_pull_requests",
|
|
497
|
+
"mcp__github__search_issues",
|
|
498
|
+
"mcp__github__get_post",
|
|
499
|
+
"mcp__trello__trelloReadCard",
|
|
500
|
+
"mcp__github__pull_request_read",
|
|
501
|
+
"mcp__jira__get_request",
|
|
502
|
+
"mcp__servicedesk__request_status",
|
|
503
|
+
],
|
|
504
|
+
)
|
|
505
|
+
def test_should_pass_a_tool_the_matcher_over_reaches_untouched(
|
|
506
|
+
monkeypatch: pytest.MonkeyPatch,
|
|
507
|
+
capsys: pytest.CaptureFixture[str],
|
|
508
|
+
tmp_path: Path,
|
|
509
|
+
tool_name: str,
|
|
510
|
+
) -> None:
|
|
511
|
+
transcript_path = acting_transcript(tmp_path, HEDGED_SENTENCE, tool_name, {})
|
|
512
|
+
exit_code, stdout_text = run_hook(monkeypatch, capsys, tool_name, {}, transcript_path)
|
|
513
|
+
assert (exit_code, stdout_text) == (0, "")
|
|
514
|
+
assert not (tmp_path / LOG_RELATIVE_PATH).exists()
|
|
515
|
+
|
|
516
|
+
|
|
517
|
+
def test_should_allow_and_log_unseen_when_the_tool_use_id_is_absent(
|
|
518
|
+
monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
|
|
519
|
+
) -> None:
|
|
520
|
+
transcript_path = write_transcript(
|
|
521
|
+
tmp_path,
|
|
522
|
+
[
|
|
523
|
+
thinking_record(ACTING_MESSAGE_ID, HEDGED_SENTENCE),
|
|
524
|
+
tool_use_record(ACTING_MESSAGE_ID, "toolu_other", "Write", WRITE_INPUT),
|
|
525
|
+
],
|
|
526
|
+
)
|
|
527
|
+
exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
|
|
528
|
+
assert (exit_code, stdout_text) == (0, "")
|
|
529
|
+
assert logged_outcomes(tmp_path) == ["reasoning_unseen"]
|
|
530
|
+
|
|
531
|
+
|
|
532
|
+
def test_should_allow_and_log_unseen_when_the_thinking_text_is_empty(
|
|
533
|
+
monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
|
|
534
|
+
) -> None:
|
|
535
|
+
transcript_path = acting_transcript(tmp_path, "", "Write", WRITE_INPUT)
|
|
536
|
+
exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
|
|
537
|
+
assert (exit_code, stdout_text) == (0, "")
|
|
538
|
+
assert logged_outcomes(tmp_path) == ["reasoning_unseen"]
|
|
539
|
+
|
|
540
|
+
|
|
541
|
+
def test_should_allow_and_log_unseen_when_the_transcript_is_missing(
|
|
542
|
+
monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
|
|
543
|
+
) -> None:
|
|
544
|
+
missing_path = tmp_path / "missing.jsonl"
|
|
545
|
+
exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, missing_path)
|
|
546
|
+
assert (exit_code, stdout_text) == (0, "")
|
|
547
|
+
assert logged_outcomes(tmp_path) == ["reasoning_unseen"]
|
|
548
|
+
|
|
549
|
+
|
|
550
|
+
def test_should_skip_malformed_lines_and_still_block(
|
|
551
|
+
monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
|
|
552
|
+
) -> None:
|
|
553
|
+
transcript_path = write_transcript(
|
|
554
|
+
tmp_path,
|
|
555
|
+
[
|
|
556
|
+
"{not json",
|
|
557
|
+
thinking_record(ACTING_MESSAGE_ID, HEDGED_SENTENCE),
|
|
558
|
+
f'{{"truncated": "{ACTING_MESSAGE_ID}',
|
|
559
|
+
"[1, 2]",
|
|
560
|
+
tool_use_record(ACTING_MESSAGE_ID, TOOL_USE_ID, "Write", WRITE_INPUT),
|
|
561
|
+
],
|
|
562
|
+
)
|
|
563
|
+
exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
|
|
564
|
+
assert exit_code == 0
|
|
565
|
+
assert json.loads(stdout_text) == expected_block("Write", HEDGED_SENTENCE)
|
|
566
|
+
|
|
567
|
+
|
|
568
|
+
def test_should_read_only_the_thinking_of_the_acting_message(
|
|
569
|
+
monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
|
|
570
|
+
) -> None:
|
|
571
|
+
transcript_path = write_transcript(
|
|
572
|
+
tmp_path,
|
|
573
|
+
[
|
|
574
|
+
thinking_record(EARLIER_MESSAGE_ID, "The file probably exists."),
|
|
575
|
+
tool_use_record(EARLIER_MESSAGE_ID, "toolu_read", "Read", {"file_path": "x"}),
|
|
576
|
+
tool_result_record("toolu_read"),
|
|
577
|
+
thinking_record(ACTING_MESSAGE_ID, CLEAN_SENTENCE),
|
|
578
|
+
tool_use_record(ACTING_MESSAGE_ID, TOOL_USE_ID, "Write", WRITE_INPUT),
|
|
579
|
+
tool_result_record(TOOL_USE_ID),
|
|
580
|
+
],
|
|
581
|
+
)
|
|
582
|
+
exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
|
|
583
|
+
assert (exit_code, stdout_text) == (0, "")
|
|
584
|
+
assert logged_outcomes(tmp_path) == ["allowed_clean"]
|
|
585
|
+
|
|
586
|
+
|
|
587
|
+
def test_should_join_thinking_records_that_share_the_message_id(
|
|
588
|
+
monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
|
|
589
|
+
) -> None:
|
|
590
|
+
transcript_path = write_transcript(
|
|
591
|
+
tmp_path,
|
|
592
|
+
[
|
|
593
|
+
thinking_record(ACTING_MESSAGE_ID, CLEAN_SENTENCE),
|
|
594
|
+
{"type": "progress", "data": {"note": "unrelated record"}},
|
|
595
|
+
thinking_record(ACTING_MESSAGE_ID, HEDGED_SENTENCE),
|
|
596
|
+
tool_use_record(ACTING_MESSAGE_ID, TOOL_USE_ID, "Write", WRITE_INPUT),
|
|
597
|
+
tool_result_record(TOOL_USE_ID),
|
|
598
|
+
],
|
|
599
|
+
)
|
|
600
|
+
exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
|
|
601
|
+
assert exit_code == 0
|
|
602
|
+
assert json.loads(stdout_text) == expected_block("Write", HEDGED_SENTENCE)
|
|
603
|
+
|
|
604
|
+
|
|
605
|
+
@pytest.mark.parametrize(
|
|
606
|
+
"thinking_text",
|
|
607
|
+
[
|
|
608
|
+
"That outcome is unlikely given the log line.",
|
|
609
|
+
"I think the next step is to write the file.",
|
|
610
|
+
"The tests should pass once this lands.",
|
|
611
|
+
],
|
|
612
|
+
)
|
|
613
|
+
def test_should_not_count_plan_words_or_partial_words_as_hedges(
|
|
614
|
+
monkeypatch: pytest.MonkeyPatch,
|
|
615
|
+
capsys: pytest.CaptureFixture[str],
|
|
616
|
+
tmp_path: Path,
|
|
617
|
+
thinking_text: str,
|
|
618
|
+
) -> None:
|
|
619
|
+
transcript_path = acting_transcript(tmp_path, thinking_text, "Write", WRITE_INPUT)
|
|
620
|
+
exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
|
|
621
|
+
assert (exit_code, stdout_text) == (0, "")
|
|
622
|
+
assert logged_outcomes(tmp_path) == ["allowed_clean"]
|
|
623
|
+
|
|
624
|
+
|
|
625
|
+
@pytest.mark.parametrize(
|
|
626
|
+
"hedge_phrase",
|
|
627
|
+
["might be", "may be", "Maybe", "appears to", "I suspect", "my theory", "not sure"],
|
|
628
|
+
)
|
|
629
|
+
def test_should_block_on_each_hedge_phrase(
|
|
630
|
+
monkeypatch: pytest.MonkeyPatch,
|
|
631
|
+
capsys: pytest.CaptureFixture[str],
|
|
632
|
+
tmp_path: Path,
|
|
633
|
+
hedge_phrase: str,
|
|
634
|
+
) -> None:
|
|
635
|
+
transcript_path = acting_transcript(
|
|
636
|
+
tmp_path, f"The cause {hedge_phrase} the stale cache.", "Write", WRITE_INPUT
|
|
637
|
+
)
|
|
638
|
+
exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
|
|
639
|
+
assert exit_code == 0
|
|
640
|
+
assert json.loads(stdout_text)["decision"] == "block"
|
|
641
|
+
|
|
642
|
+
|
|
643
|
+
def test_should_keep_the_hedge_word_when_trimming_a_long_sentence(
|
|
644
|
+
monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
|
|
645
|
+
) -> None:
|
|
646
|
+
long_sentence = ("word " * 60) + "this is probably the cause " + ("tail " * 40) + "end."
|
|
647
|
+
transcript_path = acting_transcript(tmp_path, long_sentence, "Write", WRITE_INPUT)
|
|
648
|
+
exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
|
|
649
|
+
reason = json.loads(stdout_text)["reason"]
|
|
650
|
+
quoted_sentence = reason.split('"')[1]
|
|
651
|
+
assert exit_code == 0
|
|
652
|
+
assert "probably" in quoted_sentence
|
|
653
|
+
assert quoted_sentence.startswith("...")
|
|
654
|
+
assert quoted_sentence.endswith("...")
|
|
655
|
+
assert len(quoted_sentence) <= 166
|