claude-dev-env 8.32.2 → 8.33.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (25) hide show
  1. package/hooks/advisory/pr_done_reminder.py +16 -75
  2. package/hooks/advisory/test_pr_done_reminder.py +19 -0
  3. package/hooks/blocking/test_verify_before_acting.py +655 -0
  4. package/hooks/blocking/verify_before_acting.py +323 -0
  5. package/hooks/hooks.json +10 -0
  6. package/hooks/hooks_constants/piped_pytest_blocker_constants.py +78 -6
  7. package/hooks/hooks_constants/pr_done_reminder_constants.py +0 -8
  8. package/hooks/hooks_constants/pytest_invocation.py +60 -22
  9. package/hooks/hooks_constants/shell_command_mutation.py +202 -0
  10. package/hooks/hooks_constants/shell_command_segments.py +39 -0
  11. package/hooks/hooks_constants/shell_command_wrappers.py +68 -0
  12. package/hooks/hooks_constants/test_pr_done_reminder_constants.py +0 -11
  13. package/hooks/hooks_constants/test_pytest_invocation.py +6 -0
  14. package/hooks/hooks_constants/test_shell_command_mutation.py +37 -0
  15. package/hooks/hooks_constants/test_shell_command_segments.py +29 -0
  16. package/hooks/hooks_constants/test_shell_command_wrappers.py +61 -0
  17. package/hooks/hooks_constants/verify_before_acting_constants.py +170 -0
  18. package/package.json +1 -1
  19. package/rules/agent-merges-its-own-green-pull-request.md +3 -3
  20. package/scripts/agent_merge_check.py +157 -14
  21. package/scripts/dev_env_scripts_constants/agent_merge_check_constants.py +6 -0
  22. package/scripts/policy_lint/adapter_configuration.py +4 -0
  23. package/scripts/policy_lint/config/constants.py +1 -0
  24. package/scripts/test_agent_merge_check.py +266 -0
  25. package/scripts/tests/test_adapter_configuration.py +8 -0
@@ -0,0 +1,655 @@
1
+ import io
2
+ import json
3
+ import re
4
+ import threading
5
+ from pathlib import Path
6
+
7
+ import pytest
8
+
9
+ import verify_before_acting
10
+
11
+ ACTING_MESSAGE_ID = "msg_acting"
12
+ EARLIER_MESSAGE_ID = "msg_earlier"
13
+ TOOL_USE_ID = "toolu_target"
14
+ HEDGED_SENTENCE = "The config probably lives in settings.json."
15
+ CLEAN_SENTENCE = "The config lives in settings.json, as the read above showed."
16
+ LOG_RELATIVE_PATH = Path(".claude") / "logs" / "verify-before-acting.jsonl"
17
+ WRITE_INPUT = {"file_path": "settings.json", "content": "{}"}
18
+
19
+
20
+ def thinking_record(message_id: str, thinking_text: str) -> dict[str, object]:
21
+ return {
22
+ "type": "assistant",
23
+ "message": {
24
+ "id": message_id,
25
+ "role": "assistant",
26
+ "content": [{"type": "thinking", "thinking": thinking_text, "signature": "sig"}],
27
+ },
28
+ }
29
+
30
+
31
+ def tool_use_record(
32
+ message_id: str, tool_use_id: str, tool_name: str, tool_input: dict[str, object]
33
+ ) -> dict[str, object]:
34
+ return {
35
+ "type": "assistant",
36
+ "message": {
37
+ "id": message_id,
38
+ "role": "assistant",
39
+ "content": [
40
+ {"type": "tool_use", "id": tool_use_id, "name": tool_name, "input": tool_input}
41
+ ],
42
+ },
43
+ }
44
+
45
+
46
+ def tool_result_record(tool_use_id: str) -> dict[str, object]:
47
+ return {
48
+ "type": "user",
49
+ "message": {
50
+ "role": "user",
51
+ "content": [{"type": "tool_result", "tool_use_id": tool_use_id, "content": "ok"}],
52
+ },
53
+ }
54
+
55
+
56
+ def write_transcript(directory: Path, all_lines: list[object]) -> Path:
57
+ transcript_path = directory / "transcript.jsonl"
58
+ all_texts = [
59
+ each_line if isinstance(each_line, str) else json.dumps(each_line)
60
+ for each_line in all_lines
61
+ ]
62
+ transcript_path.write_text("\n".join(all_texts) + "\n", encoding="utf-8")
63
+ return transcript_path
64
+
65
+
66
+ def acting_transcript(
67
+ directory: Path, thinking_text: str, tool_name: str, tool_input: dict[str, object]
68
+ ) -> Path:
69
+ return write_transcript(
70
+ directory,
71
+ [
72
+ thinking_record(ACTING_MESSAGE_ID, thinking_text),
73
+ tool_use_record(ACTING_MESSAGE_ID, TOOL_USE_ID, tool_name, tool_input),
74
+ ],
75
+ )
76
+
77
+
78
+ def run_hook(
79
+ monkeypatch: pytest.MonkeyPatch,
80
+ capsys: pytest.CaptureFixture[str],
81
+ tool_name: str,
82
+ tool_input: dict[str, object],
83
+ transcript_path: Path,
84
+ ) -> tuple[int, str]:
85
+ hook_input = {
86
+ "hook_event_name": "PostToolUse",
87
+ "tool_name": tool_name,
88
+ "tool_input": tool_input,
89
+ "tool_use_id": TOOL_USE_ID,
90
+ "transcript_path": str(transcript_path),
91
+ }
92
+ monkeypatch.setattr(
93
+ "sys.stdin", io.TextIOWrapper(io.BytesIO(json.dumps(hook_input).encode("utf-8")))
94
+ )
95
+ exit_code = verify_before_acting.main()
96
+ return exit_code, capsys.readouterr().out
97
+
98
+
99
+ def logged_outcomes(home_directory: Path) -> list[str]:
100
+ log_path = home_directory / LOG_RELATIVE_PATH
101
+ if not log_path.exists():
102
+ return []
103
+ all_records = [
104
+ json.loads(each_line)
105
+ for each_line in log_path.read_text(encoding="utf-8").splitlines()
106
+ if each_line
107
+ ]
108
+ return [each_record["outcome"] for each_record in all_records]
109
+
110
+
111
+ def expected_block(tool_name: str, quoted_sentence: str) -> dict[str, object]:
112
+ return {
113
+ "decision": "block",
114
+ "reason": (
115
+ f'The reasoning behind this {tool_name} call hedges: "{quoted_sentence}" '
116
+ "Check that claim now with a read-only tool, then continue. "
117
+ "If the check contradicts it, undo this change first."
118
+ ),
119
+ "hookSpecificOutput": {"hookEventName": "PostToolUse"},
120
+ }
121
+
122
+
123
+ @pytest.fixture(autouse=True)
124
+ def isolated_home(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> None:
125
+ monkeypatch.setenv("HOME", str(tmp_path))
126
+ monkeypatch.setenv("USERPROFILE", str(tmp_path))
127
+ monkeypatch.setattr(verify_before_acting, "TRANSCRIPT_POLL_LIMIT_SECONDS", 0.3)
128
+
129
+
130
+ def test_should_block_a_hedged_write_whose_record_lands_after_the_hook_starts(
131
+ monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
132
+ ) -> None:
133
+ transcript_path = write_transcript(tmp_path, [thinking_record(EARLIER_MESSAGE_ID, "Start.")])
134
+ late_lines = "".join(
135
+ json.dumps(each_record) + "\n"
136
+ for each_record in (
137
+ thinking_record(ACTING_MESSAGE_ID, HEDGED_SENTENCE),
138
+ tool_use_record(ACTING_MESSAGE_ID, TOOL_USE_ID, "Write", WRITE_INPUT),
139
+ )
140
+ )
141
+
142
+ def append_late_lines() -> None:
143
+ with transcript_path.open("a", encoding="utf-8") as transcript_file:
144
+ transcript_file.write(late_lines)
145
+
146
+ writer = threading.Timer(0.15, append_late_lines)
147
+ writer.start()
148
+ exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
149
+ writer.join()
150
+
151
+ assert exit_code == 0
152
+ assert json.loads(stdout_text) == expected_block("Write", HEDGED_SENTENCE)
153
+ assert logged_outcomes(tmp_path) == ["blocked"]
154
+
155
+
156
+ def test_should_allow_and_log_unseen_when_the_record_never_lands(
157
+ monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
158
+ ) -> None:
159
+ transcript_path = write_transcript(tmp_path, [thinking_record(EARLIER_MESSAGE_ID, "Start.")])
160
+ exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
161
+ assert (exit_code, stdout_text) == (0, "")
162
+ assert logged_outcomes(tmp_path) == ["reasoning_unseen"]
163
+
164
+
165
+ def test_should_block_a_write_after_hedged_reasoning(
166
+ monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
167
+ ) -> None:
168
+ transcript_path = acting_transcript(
169
+ tmp_path, f"I need the config.\n{HEDGED_SENTENCE} Writing it now.", "Write", WRITE_INPUT
170
+ )
171
+ exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
172
+ assert exit_code == 0
173
+ assert json.loads(stdout_text) == expected_block("Write", HEDGED_SENTENCE)
174
+ assert logged_outcomes(tmp_path) == ["blocked"]
175
+
176
+
177
+ def test_should_log_the_quoted_hedge_sentence_on_a_block(
178
+ monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
179
+ ) -> None:
180
+ transcript_path = acting_transcript(tmp_path, HEDGED_SENTENCE, "Write", WRITE_INPUT)
181
+ run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
182
+ log_line = (tmp_path / LOG_RELATIVE_PATH).read_text(encoding="utf-8").splitlines()[0]
183
+ log_record = json.loads(log_line)
184
+ assert log_record["tool_name"] == "Write"
185
+ assert log_record["tool_use_id"] == TOOL_USE_ID
186
+ assert log_record["hedge_sentence"] == HEDGED_SENTENCE
187
+ assert log_record["timestamp"]
188
+
189
+
190
+ def test_should_allow_a_write_after_clean_reasoning(
191
+ monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
192
+ ) -> None:
193
+ transcript_path = acting_transcript(tmp_path, CLEAN_SENTENCE, "Write", WRITE_INPUT)
194
+ exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
195
+ assert (exit_code, stdout_text) == (0, "")
196
+ assert logged_outcomes(tmp_path) == ["allowed_clean"]
197
+
198
+
199
+ @pytest.mark.parametrize(
200
+ ("tool_name", "command"),
201
+ [
202
+ ("Bash", "gh pr view 12 --json state"),
203
+ ("PowerShell", "Get-Content settings.json"),
204
+ ("PowerShell", "Test-Path settings.json"),
205
+ ("Bash", "gh api repos/o/r/pulls/12"),
206
+ ("Bash", "git status 2>&1"),
207
+ ("Bash", "git log --oneline > /dev/null"),
208
+ ("Bash", "cat notes.txt"),
209
+ ("Bash", "cat touch.txt"),
210
+ ("Bash", "ls -la"),
211
+ ("Bash", "ls mkdir_notes"),
212
+ ("Bash", "grep -rn rm src"),
213
+ ("Bash", 'grep -n "cp " notes.md'),
214
+ ("Bash", "git status"),
215
+ ("Bash", "git diff"),
216
+ ("Bash", "git diff -- cp.py"),
217
+ ("Bash", "gh pr view 12"),
218
+ ("Bash", "sed -n 1,5p notes.txt"),
219
+ ("Bash", "sed --silent 1p notes.txt"),
220
+ ("Bash", "git log --oneline -- rm.py"),
221
+ ("Bash", "python pull_request.py --help"),
222
+ ("Bash", "awk '$1 > 5' notes.txt"),
223
+ ("Bash", "jq '.count > 3' data.json"),
224
+ ("Bash", "gh pr list --json number --jq 'map(select(.number > 5))'"),
225
+ ("Bash", 'python -c "print(1 > 0)"'),
226
+ ("Bash", "grep -c '>' notes.txt"),
227
+ ("Bash", "git log --format='%h > %s'"),
228
+ ("Bash", "gh api -X GET search/issues -f q=repo:o/r"),
229
+ ("Bash", "gh api --method GET repos/o/r/pulls -f state=open"),
230
+ ("Bash", "git --no-pager log"),
231
+ ("Bash", "git --git-dir=/repo/.git log --oneline"),
232
+ ("Bash", "git -C /repo stash list"),
233
+ ("Bash", "git -c core.pager=cat diff"),
234
+ ("Bash", 'pwsh -NoProfile -Command "git status"'),
235
+ ("PowerShell", "Get-Content x.txt > $null"),
236
+ ("Bash", "rg 'gh pr merge' docs"),
237
+ ("Bash", 'grep -rn "gh issue comment" docs'),
238
+ ("Bash", "rg 'gh api repos/o/r/issues -f title=x' docs"),
239
+ ("Bash", "rg 'gh workflow run' .github"),
240
+ ("Bash", "rg 'Remove-Item' docs"),
241
+ ("PowerShell", "Select-String -Pattern 'Set-Content' -Path notes.md"),
242
+ ("Bash", 'grep "a|cp b" notes.md'),
243
+ ("Bash", "grep 'x; rm y' notes.md"),
244
+ ("Bash", "rg 'sed -i' docs"),
245
+ ("Bash", "rg 'pull_request.py create' docs"),
246
+ ("Bash", "sudo cat notes.txt"),
247
+ ("Bash", 'bash -c "git status"'),
248
+ ("Bash", "sh -c ls"),
249
+ ("Bash", "bash -lc 'git log --oneline'"),
250
+ ("Bash", "env VAR=x git status"),
251
+ ("Bash", "echo HEAD | xargs git show"),
252
+ ("Bash", "pwsh -Command git status"),
253
+ ("Bash", "pwsh -Command Get-Content notes.md"),
254
+ ("Bash", "env -u HOME git status"),
255
+ ("Bash", "timeout --signal KILL 5 git status"),
256
+ ("Bash", "printf x | xargs -I {} cat {}"),
257
+ ("Bash", "printf x | xargs -n 1 git show"),
258
+ ("PowerShell", "Get-Command Remove-Item"),
259
+ ("PowerShell", "Get-Help Set-Content -Full"),
260
+ ("Bash", "echo hi >&-"),
261
+ ("Bash", "echo hi 1>&2"),
262
+ ("Bash", "git.exe status"),
263
+ ("Bash", "gh pr view 8 --json reviews"),
264
+ ],
265
+ )
266
+ def test_should_pass_a_read_only_command_untouched(
267
+ monkeypatch: pytest.MonkeyPatch,
268
+ capsys: pytest.CaptureFixture[str],
269
+ tmp_path: Path,
270
+ tool_name: str,
271
+ command: str,
272
+ ) -> None:
273
+ tool_input = {"command": command}
274
+ transcript_path = acting_transcript(tmp_path, HEDGED_SENTENCE, tool_name, tool_input)
275
+ exit_code, stdout_text = run_hook(monkeypatch, capsys, tool_name, tool_input, transcript_path)
276
+ assert (exit_code, stdout_text) == (0, "")
277
+ assert not (tmp_path / LOG_RELATIVE_PATH).exists()
278
+
279
+
280
+ @pytest.mark.parametrize(
281
+ ("tool_name", "command"),
282
+ [
283
+ ("Bash", "git push origin feat/x"),
284
+ ("Bash", 'git -C "C:/repo dir" commit -F message.txt'),
285
+ ("Bash", "gh pr merge 12 --squash"),
286
+ ("Bash", "gh issue comment 4 --body-file body.md"),
287
+ ("Bash", "gh api repos/o/r/issues/4/comments -X POST"),
288
+ ("Bash", "gh api repos/o/r/pulls/12 --method PATCH"),
289
+ ("Bash", "gh api repos/o/r/issues -f title=x"),
290
+ ("Bash", "gh run rerun 99"),
291
+ ("Bash", "gh workflow run ci.yml"),
292
+ ("PowerShell", "Remove-Item -Force C:/scratch/x.txt"),
293
+ ("PowerShell", "set-content x.txt 'y'"),
294
+ ("Bash", "echo hi > notes.txt"),
295
+ ("Bash", "rm notes.txt"),
296
+ ("Bash", "rm -rf build"),
297
+ ("Bash", "rmdir out"),
298
+ ("Bash", "unlink link.txt"),
299
+ ("Bash", "touch notes.txt"),
300
+ ("Bash", "cp source.txt notes.txt"),
301
+ ("Bash", "mv old.txt new.txt"),
302
+ ("Bash", "mkdir -p out"),
303
+ ("Bash", "echo hi | tee notes.txt"),
304
+ ("Bash", "sed -i 's/a/b/' notes.txt"),
305
+ ("Bash", "sed -i.bak 's/a/b/' notes.txt"),
306
+ ("Bash", "sed -e 's/a/b/' -i notes.txt"),
307
+ ("Bash", "sed --in-place 's/a/b/' notes.txt"),
308
+ ("Bash", "ln notes.txt link.txt"),
309
+ ("Bash", "ln -s target link"),
310
+ ("Bash", "ln -sf target link"),
311
+ ("Bash", "git status && rm notes.txt"),
312
+ ("Bash", "sudo rm notes.txt"),
313
+ ("Bash", "find . -name '*.tmp' | xargs rm"),
314
+ ("Bash", "/bin/rm notes.txt"),
315
+ ("Bash", "python ~/.agents/skills/pull-request/scripts/pull_request.py create --repo o/r"),
316
+ ("Bash", "python pull_request.py edit --repo o/r --number 12 --body-file body.md"),
317
+ ("Bash", "python pull_request.py comment --repo o/r --number 12 --body-file body.md"),
318
+ ("Bash", "python pull_request.py review --repo o/r --number 12 --event approve"),
319
+ ("Bash", 'pwsh -NoProfile -Command "git push origin HEAD"'),
320
+ ("Bash", "cd repo && git checkout -b feat/x"),
321
+ ("Bash", "echo hi>notes.txt"),
322
+ ("Bash", "echo hi >> notes.txt"),
323
+ ("Bash", "make 2> errors.log"),
324
+ ("Bash", "make &> out.log"),
325
+ ("PowerShell", "rm C:/scratch/x.txt"),
326
+ ("PowerShell", "mkdir out"),
327
+ ("Bash", "pwsh -Command git push origin HEAD"),
328
+ ("Bash", "pwsh -Command git push origin HEAD; echo done"),
329
+ ("Bash", 'pwsh -NoProfile -Command "Get-ChildItem *.tmp | Remove-Item"'),
330
+ ("PowerShell", "Get-ChildItem *.tmp | ForEach-Object { Remove-Item $_ }"),
331
+ ("Bash", "bash -c 'gh pr merge 12 --squash'"),
332
+ ("Bash", "sudo gh api repos/o/r/issues -X POST"),
333
+ ("Bash", "gh api repos/o/r/issues -XPOST"),
334
+ ("Bash", "gh api repos/o/r/pulls/12 --method=PATCH"),
335
+ ("Bash", "printf x | xargs -r rm"),
336
+ ("Bash", "printf x | xargs -I {} rm {}"),
337
+ ("Bash", "printf x | xargs -i rm {}"),
338
+ ("Bash", "printf x | xargs -n 1 rm"),
339
+ ("Bash", "find . -name '*.tmp' -print0 | xargs -0 rm"),
340
+ ("Bash", "gh api repos/o/r/issues --input payload.json"),
341
+ ("Bash", "gh api repos/o/r/issues --input=payload.json"),
342
+ ("Bash", "git.exe push origin HEAD"),
343
+ ("Bash", "gh.exe pr merge 12 --squash"),
344
+ ("PowerShell", "& 'C:/Program Files/Git/cmd/git.exe' commit -F message.txt"),
345
+ ("Bash", "echo hi > 1"),
346
+ ("Bash", "echo hi > -"),
347
+ ("Bash", "echo hi >| notes.txt"),
348
+ ("Bash", "echo hi>|notes.txt"),
349
+ ("Bash", "gh pr review 8 --approve"),
350
+ ("PowerShell", "Get-ChildItem *.tmp | % {Remove-Item $_}"),
351
+ ],
352
+ )
353
+ def test_should_block_a_mutating_command_after_hedged_reasoning(
354
+ monkeypatch: pytest.MonkeyPatch,
355
+ capsys: pytest.CaptureFixture[str],
356
+ tmp_path: Path,
357
+ tool_name: str,
358
+ command: str,
359
+ ) -> None:
360
+ tool_input = {"command": command}
361
+ transcript_path = acting_transcript(tmp_path, HEDGED_SENTENCE, tool_name, tool_input)
362
+ exit_code, stdout_text = run_hook(monkeypatch, capsys, tool_name, tool_input, transcript_path)
363
+ assert exit_code == 0
364
+ assert json.loads(stdout_text) == expected_block(tool_name, HEDGED_SENTENCE)
365
+
366
+
367
+ @pytest.mark.parametrize(
368
+ "global_options",
369
+ [
370
+ "--git-dir=/repo/.git",
371
+ "--git-dir /repo/.git",
372
+ "--work-tree=/repo",
373
+ "--work-tree /repo",
374
+ "-C /repo",
375
+ "-c user.name=x",
376
+ "--no-pager",
377
+ "--namespace=n",
378
+ "--namespace n",
379
+ "-c user.name=x -C /repo",
380
+ ],
381
+ )
382
+ @pytest.mark.parametrize("subcommand", ["push origin HEAD", "commit -F message.txt"])
383
+ def test_should_block_a_git_change_behind_global_options(
384
+ monkeypatch: pytest.MonkeyPatch,
385
+ capsys: pytest.CaptureFixture[str],
386
+ tmp_path: Path,
387
+ global_options: str,
388
+ subcommand: str,
389
+ ) -> None:
390
+ tool_input = {"command": f"git {global_options} {subcommand}"}
391
+ transcript_path = acting_transcript(tmp_path, HEDGED_SENTENCE, "Bash", tool_input)
392
+ exit_code, stdout_text = run_hook(monkeypatch, capsys, "Bash", tool_input, transcript_path)
393
+ assert exit_code == 0
394
+ assert json.loads(stdout_text) == expected_block("Bash", HEDGED_SENTENCE)
395
+
396
+
397
+ @pytest.mark.parametrize(
398
+ "command_template",
399
+ [
400
+ "sudo git {subcommand}",
401
+ "sudo -u root git {subcommand}",
402
+ 'bash -c "git {subcommand}"',
403
+ "sh -c 'git {subcommand}'",
404
+ "bash -lc 'git {subcommand}'",
405
+ "env VAR=x git {subcommand}",
406
+ "echo origin | xargs git {subcommand}",
407
+ "pwsh -Command git {subcommand}",
408
+ "sudo bash -c 'git {subcommand}'",
409
+ "env -u HOME git {subcommand}",
410
+ "env -C /repo git {subcommand}",
411
+ "timeout --signal KILL 5 git {subcommand}",
412
+ "timeout -k 2 5 git {subcommand}",
413
+ "timeout 5 git {subcommand}",
414
+ "echo origin | xargs -n 1 git {subcommand}",
415
+ "echo origin | xargs -r git {subcommand}",
416
+ "echo origin | xargs -I {{}} git {subcommand}",
417
+ "echo origin | xargs -0 -P 4 git {subcommand}",
418
+ ],
419
+ )
420
+ @pytest.mark.parametrize("subcommand", ["push origin HEAD", "commit -F message.txt"])
421
+ def test_should_block_a_git_change_behind_a_wrapper(
422
+ monkeypatch: pytest.MonkeyPatch,
423
+ capsys: pytest.CaptureFixture[str],
424
+ tmp_path: Path,
425
+ command_template: str,
426
+ subcommand: str,
427
+ ) -> None:
428
+ tool_input = {"command": command_template.format(subcommand=subcommand)}
429
+ transcript_path = acting_transcript(tmp_path, HEDGED_SENTENCE, "Bash", tool_input)
430
+ exit_code, stdout_text = run_hook(monkeypatch, capsys, "Bash", tool_input, transcript_path)
431
+ assert exit_code == 0
432
+ assert json.loads(stdout_text) == expected_block("Bash", HEDGED_SENTENCE)
433
+
434
+
435
+ @pytest.mark.parametrize(
436
+ "tool_name",
437
+ [
438
+ "Edit",
439
+ "MultiEdit",
440
+ "NotebookEdit",
441
+ "Agent",
442
+ "Task",
443
+ "apply_patch",
444
+ "mcp__gmail__send_message",
445
+ "mcp__github__issue_write",
446
+ "mcp__github__sub_issue_write",
447
+ "mcp__github__pull_request_review_write",
448
+ "mcp__github__add_issue_comment",
449
+ "mcp__atlassian__createJiraIssue",
450
+ "mcp__trello__trelloWriteCard",
451
+ "mcp__github__request_copilot_review",
452
+ ],
453
+ )
454
+ def test_should_block_each_always_mutating_tool_after_hedged_reasoning(
455
+ monkeypatch: pytest.MonkeyPatch,
456
+ capsys: pytest.CaptureFixture[str],
457
+ tmp_path: Path,
458
+ tool_name: str,
459
+ ) -> None:
460
+ transcript_path = acting_transcript(tmp_path, HEDGED_SENTENCE, tool_name, {})
461
+ exit_code, stdout_text = run_hook(monkeypatch, capsys, tool_name, {}, transcript_path)
462
+ assert exit_code == 0
463
+ assert json.loads(stdout_text)["decision"] == "block"
464
+
465
+
466
+ def test_should_register_a_matcher_covering_every_mutating_tool_name() -> None:
467
+ hooks_configuration = json.loads(
468
+ (Path(__file__).resolve().parent.parent / "hooks.json").read_text(encoding="utf-8")
469
+ )
470
+ all_matchers = [
471
+ each_entry["matcher"]
472
+ for each_entry in hooks_configuration["hooks"]["PostToolUse"]
473
+ if any(
474
+ "verify_before_acting.py" in each_hook["command"] for each_hook in each_entry["hooks"]
475
+ )
476
+ ]
477
+ assert len(all_matchers) == 1
478
+ matcher_pattern = re.compile(all_matchers[0])
479
+ all_expected_names = (
480
+ verify_before_acting.ALL_ALWAYS_MUTATING_TOOL_NAMES
481
+ | verify_before_acting.ALL_SHELL_TOOL_NAMES
482
+ )
483
+ assert {
484
+ each_name for each_name in all_expected_names if not matcher_pattern.fullmatch(each_name)
485
+ } == set()
486
+
487
+
488
+ @pytest.mark.parametrize(
489
+ "tool_name",
490
+ [
491
+ "TodoWrite",
492
+ "TaskUpdate",
493
+ "Read",
494
+ "mcp__gmail__get_thread",
495
+ "mcp__github__issue_read",
496
+ "mcp__github__list_pull_requests",
497
+ "mcp__github__search_issues",
498
+ "mcp__github__get_post",
499
+ "mcp__trello__trelloReadCard",
500
+ "mcp__github__pull_request_read",
501
+ "mcp__jira__get_request",
502
+ "mcp__servicedesk__request_status",
503
+ ],
504
+ )
505
+ def test_should_pass_a_tool_the_matcher_over_reaches_untouched(
506
+ monkeypatch: pytest.MonkeyPatch,
507
+ capsys: pytest.CaptureFixture[str],
508
+ tmp_path: Path,
509
+ tool_name: str,
510
+ ) -> None:
511
+ transcript_path = acting_transcript(tmp_path, HEDGED_SENTENCE, tool_name, {})
512
+ exit_code, stdout_text = run_hook(monkeypatch, capsys, tool_name, {}, transcript_path)
513
+ assert (exit_code, stdout_text) == (0, "")
514
+ assert not (tmp_path / LOG_RELATIVE_PATH).exists()
515
+
516
+
517
+ def test_should_allow_and_log_unseen_when_the_tool_use_id_is_absent(
518
+ monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
519
+ ) -> None:
520
+ transcript_path = write_transcript(
521
+ tmp_path,
522
+ [
523
+ thinking_record(ACTING_MESSAGE_ID, HEDGED_SENTENCE),
524
+ tool_use_record(ACTING_MESSAGE_ID, "toolu_other", "Write", WRITE_INPUT),
525
+ ],
526
+ )
527
+ exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
528
+ assert (exit_code, stdout_text) == (0, "")
529
+ assert logged_outcomes(tmp_path) == ["reasoning_unseen"]
530
+
531
+
532
+ def test_should_allow_and_log_unseen_when_the_thinking_text_is_empty(
533
+ monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
534
+ ) -> None:
535
+ transcript_path = acting_transcript(tmp_path, "", "Write", WRITE_INPUT)
536
+ exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
537
+ assert (exit_code, stdout_text) == (0, "")
538
+ assert logged_outcomes(tmp_path) == ["reasoning_unseen"]
539
+
540
+
541
+ def test_should_allow_and_log_unseen_when_the_transcript_is_missing(
542
+ monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
543
+ ) -> None:
544
+ missing_path = tmp_path / "missing.jsonl"
545
+ exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, missing_path)
546
+ assert (exit_code, stdout_text) == (0, "")
547
+ assert logged_outcomes(tmp_path) == ["reasoning_unseen"]
548
+
549
+
550
+ def test_should_skip_malformed_lines_and_still_block(
551
+ monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
552
+ ) -> None:
553
+ transcript_path = write_transcript(
554
+ tmp_path,
555
+ [
556
+ "{not json",
557
+ thinking_record(ACTING_MESSAGE_ID, HEDGED_SENTENCE),
558
+ f'{{"truncated": "{ACTING_MESSAGE_ID}',
559
+ "[1, 2]",
560
+ tool_use_record(ACTING_MESSAGE_ID, TOOL_USE_ID, "Write", WRITE_INPUT),
561
+ ],
562
+ )
563
+ exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
564
+ assert exit_code == 0
565
+ assert json.loads(stdout_text) == expected_block("Write", HEDGED_SENTENCE)
566
+
567
+
568
+ def test_should_read_only_the_thinking_of_the_acting_message(
569
+ monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
570
+ ) -> None:
571
+ transcript_path = write_transcript(
572
+ tmp_path,
573
+ [
574
+ thinking_record(EARLIER_MESSAGE_ID, "The file probably exists."),
575
+ tool_use_record(EARLIER_MESSAGE_ID, "toolu_read", "Read", {"file_path": "x"}),
576
+ tool_result_record("toolu_read"),
577
+ thinking_record(ACTING_MESSAGE_ID, CLEAN_SENTENCE),
578
+ tool_use_record(ACTING_MESSAGE_ID, TOOL_USE_ID, "Write", WRITE_INPUT),
579
+ tool_result_record(TOOL_USE_ID),
580
+ ],
581
+ )
582
+ exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
583
+ assert (exit_code, stdout_text) == (0, "")
584
+ assert logged_outcomes(tmp_path) == ["allowed_clean"]
585
+
586
+
587
+ def test_should_join_thinking_records_that_share_the_message_id(
588
+ monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
589
+ ) -> None:
590
+ transcript_path = write_transcript(
591
+ tmp_path,
592
+ [
593
+ thinking_record(ACTING_MESSAGE_ID, CLEAN_SENTENCE),
594
+ {"type": "progress", "data": {"note": "unrelated record"}},
595
+ thinking_record(ACTING_MESSAGE_ID, HEDGED_SENTENCE),
596
+ tool_use_record(ACTING_MESSAGE_ID, TOOL_USE_ID, "Write", WRITE_INPUT),
597
+ tool_result_record(TOOL_USE_ID),
598
+ ],
599
+ )
600
+ exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
601
+ assert exit_code == 0
602
+ assert json.loads(stdout_text) == expected_block("Write", HEDGED_SENTENCE)
603
+
604
+
605
+ @pytest.mark.parametrize(
606
+ "thinking_text",
607
+ [
608
+ "That outcome is unlikely given the log line.",
609
+ "I think the next step is to write the file.",
610
+ "The tests should pass once this lands.",
611
+ ],
612
+ )
613
+ def test_should_not_count_plan_words_or_partial_words_as_hedges(
614
+ monkeypatch: pytest.MonkeyPatch,
615
+ capsys: pytest.CaptureFixture[str],
616
+ tmp_path: Path,
617
+ thinking_text: str,
618
+ ) -> None:
619
+ transcript_path = acting_transcript(tmp_path, thinking_text, "Write", WRITE_INPUT)
620
+ exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
621
+ assert (exit_code, stdout_text) == (0, "")
622
+ assert logged_outcomes(tmp_path) == ["allowed_clean"]
623
+
624
+
625
+ @pytest.mark.parametrize(
626
+ "hedge_phrase",
627
+ ["might be", "may be", "Maybe", "appears to", "I suspect", "my theory", "not sure"],
628
+ )
629
+ def test_should_block_on_each_hedge_phrase(
630
+ monkeypatch: pytest.MonkeyPatch,
631
+ capsys: pytest.CaptureFixture[str],
632
+ tmp_path: Path,
633
+ hedge_phrase: str,
634
+ ) -> None:
635
+ transcript_path = acting_transcript(
636
+ tmp_path, f"The cause {hedge_phrase} the stale cache.", "Write", WRITE_INPUT
637
+ )
638
+ exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
639
+ assert exit_code == 0
640
+ assert json.loads(stdout_text)["decision"] == "block"
641
+
642
+
643
+ def test_should_keep_the_hedge_word_when_trimming_a_long_sentence(
644
+ monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], tmp_path: Path
645
+ ) -> None:
646
+ long_sentence = ("word " * 60) + "this is probably the cause " + ("tail " * 40) + "end."
647
+ transcript_path = acting_transcript(tmp_path, long_sentence, "Write", WRITE_INPUT)
648
+ exit_code, stdout_text = run_hook(monkeypatch, capsys, "Write", WRITE_INPUT, transcript_path)
649
+ reason = json.loads(stdout_text)["reason"]
650
+ quoted_sentence = reason.split('"')[1]
651
+ assert exit_code == 0
652
+ assert "probably" in quoted_sentence
653
+ assert quoted_sentence.startswith("...")
654
+ assert quoted_sentence.endswith("...")
655
+ assert len(quoted_sentence) <= 166