judgetap 0.0.2.dev26__tar.gz → 0.0.2.dev27__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/PKG-INFO +2 -2
  2. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/README.md +1 -1
  3. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/docs/SPEC.md +1 -1
  4. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/pyproject.toml +1 -1
  5. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/__init__.py +1 -1
  6. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/guard/install.py +13 -2
  7. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/guard/loop.py +87 -3
  8. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/guard/stop.py +8 -0
  9. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_guard_loop.py +163 -8
  10. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_guard_stop.py +37 -0
  11. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/.github/workflows/ci.yml +0 -0
  12. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/.github/workflows/demo.yml +0 -0
  13. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/.github/workflows/release.yml +0 -0
  14. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/.gitignore +0 -0
  15. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/.python-version +0 -0
  16. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/.release-please-manifest.json +0 -0
  17. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/CONTRIBUTING.md +0 -0
  18. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/LICENSE +0 -0
  19. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/docs/demo.tape +0 -0
  20. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/release-please-config.json +0 -0
  21. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/_compat.py +0 -0
  22. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/api.py +0 -0
  23. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/cascade.py +0 -0
  24. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/cli.py +0 -0
  25. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/dashboard/__init__.py +0 -0
  26. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/dashboard/data.py +0 -0
  27. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/dashboard/page.html +0 -0
  28. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/dashboard/server.py +0 -0
  29. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/decision_log.py +0 -0
  30. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/engine.py +0 -0
  31. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/engines/__init__.py +0 -0
  32. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/engines/agentjev.py +0 -0
  33. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/engines/jev.py +0 -0
  34. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/engines/laya.py +0 -0
  35. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/engines/llm.py +0 -0
  36. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/errors.py +0 -0
  37. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/evaluate.py +0 -0
  38. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/guard/__init__.py +0 -0
  39. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/guard/core.py +0 -0
  40. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/guard/hook.py +0 -0
  41. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/guard/rules.py +0 -0
  42. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/py.typed +0 -0
  43. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/secrets.py +0 -0
  44. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/testing.py +0 -0
  45. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/types.py +0 -0
  46. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_api.py +0 -0
  47. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_call_accounting.py +0 -0
  48. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_calls.py +0 -0
  49. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_cascade.py +0 -0
  50. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_dashboard.py +0 -0
  51. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_decision_log.py +0 -0
  52. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_engine_jev.py +0 -0
  53. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_engine_llm.py +0 -0
  54. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_engine_local.py +0 -0
  55. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_evaluate.py +0 -0
  56. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_guard_agents.py +0 -0
  57. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_guard_core.py +0 -0
  58. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_guard_hook.py +0 -0
  59. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_guard_rules.py +0 -0
  60. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_questions.py +0 -0
  61. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_secrets.py +0 -0
  62. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/uv.lock +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: judgetap
3
- Version: 0.0.2.dev26
3
+ Version: 0.0.2.dev27
4
4
  Summary: Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development.
5
5
  Project-URL: Homepage, https://github.com/mergesafe-ai/judgetap
6
6
  Author-email: Omer Bar-Ness <omer@zsquared.io>
@@ -80,7 +80,7 @@ On a CPU-only Linux box, install the CPU PyTorch wheel before `judgetap[laya]` (
80
80
 
81
81
  ## Guard details
82
82
 
83
- **Loop detection (Claude Code).** A `PostToolUse` hook (`judgetap guard post`, added by `guard install --for claude-code`) notices when the same command or edit fails the same way three times within eight actions and adds a note asking the agent to re-plan. It never blocks, uses no model, and stores only a redacted action and an error hash per session. Cursor and Codex: not yet.
83
+ **Loop detection (Claude Code).** `PostToolUse` and `PostToolUseFailure` hooks (`judgetap guard post`, added by `guard install --for claude-code`) notices when the same command or edit fails the same way three times within eight actions and adds a note asking the agent to re-plan. It never blocks, uses no model, and stores only a redacted action and an error hash per session. Cursor and Codex: not yet.
84
84
 
85
85
  - **Agents:** Claude Code (shell, writes and edits), Cursor and Codex (shell only; their hooks don't expose writes and edits).
86
86
  - **Engine:** `judgetap guard install` uses one you already have (`$JUDGETAP_ENGINE`, a `TYPESAFE_API_KEY`, or a local AgentJev) and saves it in `~/.judgetap/guard.toml`, because agents often run hooks without your shell's environment.
@@ -63,7 +63,7 @@ On a CPU-only Linux box, install the CPU PyTorch wheel before `judgetap[laya]` (
63
63
 
64
64
  ## Guard details
65
65
 
66
- **Loop detection (Claude Code).** A `PostToolUse` hook (`judgetap guard post`, added by `guard install --for claude-code`) notices when the same command or edit fails the same way three times within eight actions and adds a note asking the agent to re-plan. It never blocks, uses no model, and stores only a redacted action and an error hash per session. Cursor and Codex: not yet.
66
+ **Loop detection (Claude Code).** `PostToolUse` and `PostToolUseFailure` hooks (`judgetap guard post`, added by `guard install --for claude-code`) notices when the same command or edit fails the same way three times within eight actions and adds a note asking the agent to re-plan. It never blocks, uses no model, and stores only a redacted action and an error hash per session. Cursor and Codex: not yet.
67
67
 
68
68
  - **Agents:** Claude Code (shell, writes and edits), Cursor and Codex (shell only; their hooks don't expose writes and edits).
69
69
  - **Engine:** `judgetap guard install` uses one you already have (`$JUDGETAP_ENGINE`, a `TYPESAFE_API_KEY`, or a local AgentJev) and saves it in `~/.judgetap/guard.toml`, because agents often run hooks without your shell's environment.
@@ -67,7 +67,7 @@ A pre-action hook for coding agents, built on the core.
67
67
  - **Outcomes**: allow (silent), hold (block with a reason the agent reads and re-plans from), ask (escalate to the user). Holds should be rare; the target is under 5 per 1,000 calls.
68
68
  - **Fails safe and visibly**: Claude Code treats a crashing hook as non-blocking, so the guard catches its own errors, applies the rules layer alone, and says so.
69
69
  - **Log**: every decision to a local JSONL, so `judgetap guard stats` can report holds and cost. Marking a hold as a false alarm comes with the dashboard (#10).
70
- - **Loop detection** (Claude Code `PostToolUse`, no model): the same action failing with the same error (numbers ignored) 3 times in the last 8 actions adds `additionalContext` telling the agent to re-plan; a success of that action resets the count. Never blocks. Per-session ring buffer of 20 redacted actions and error hashes in `~/.judgetap/sessions/`, 0600. Logged as layer `loop`, outcome `note`.
70
+ - **Loop detection** (Claude Code `PostToolUseFailure` for failures, `PostToolUse` for successes that reset a streak; no model): the same action failing with the same error (numbers ignored) 3 times in the last 8 actions adds `additionalContext` telling the agent to re-plan; a success of that action resets the count. Never blocks. Per-session ring buffer of 20 redacted actions and error hashes in `~/.judgetap/sessions/`, 0600. Logged as layer `loop`, outcome `note`.
71
71
 
72
72
  ## Decided: the guard's default engine (#8)
73
73
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "judgetap"
3
- version = "0.0.2.dev26"
3
+ version = "0.0.2.dev27"
4
4
  description = "Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development."
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -23,7 +23,7 @@ from judgetap.errors import (
23
23
  )
24
24
  from judgetap.types import Decision, Question
25
25
 
26
- __version__ = "0.0.2.dev26" # x-release-please-version
26
+ __version__ = "0.0.2.dev27" # x-release-please-version
27
27
 
28
28
  __all__ = [
29
29
  "Cascade",
@@ -89,7 +89,14 @@ def _event(agent: str) -> str:
89
89
 
90
90
  def _events(agent: str, with_stop: bool = False) -> list[str]:
91
91
  if agent == "claude-code":
92
- return ["PreToolUse", "PostToolUse", *(["Stop"] if with_stop else [])]
92
+ # Loop detection needs both: failures arrive only on
93
+ # PostToolUseFailure, successes (which end a streak) on PostToolUse.
94
+ return [
95
+ "PreToolUse",
96
+ "PostToolUse",
97
+ "PostToolUseFailure",
98
+ *(["Stop"] if with_stop else []),
99
+ ]
93
100
  return [_event(agent)]
94
101
 
95
102
 
@@ -98,7 +105,11 @@ def _entry(agent: str, event: str) -> dict:
98
105
  return {"command": hook_command(agent)}
99
106
  if event == "Stop": # Stop takes no matcher
100
107
  return {"hooks": [{"type": "command", "command": STOP_COMMAND}]}
101
- command = POST_COMMAND if event == "PostToolUse" else hook_command(agent)
108
+ command = (
109
+ POST_COMMAND
110
+ if event in ("PostToolUse", "PostToolUseFailure")
111
+ else hook_command(agent)
112
+ )
102
113
  # Codex's PreToolUse fires for shell only today; the matcher says so.
103
114
  matcher = "^(exec_command|shell|Bash)$" if agent == "codex" else MATCHER
104
115
  return {"matcher": matcher, "hooks": [{"type": "command", "command": command}]}
@@ -13,6 +13,7 @@ import json
13
13
  import os
14
14
  import re
15
15
  import sys
16
+ import time
16
17
  import uuid
17
18
  from contextlib import contextmanager
18
19
  from datetime import UTC, datetime
@@ -50,10 +51,20 @@ EXIT_PREFIX = re.compile(r"^\s*exit code[: ]\s*(-?\d+)", re.IGNORECASE)
50
51
 
51
52
 
52
53
  def failure(payload: dict[str, Any]) -> str | None:
53
- """A short hash of the error, or None when the call succeeded."""
54
+ """A short hash of the error, or None when the call succeeded.
55
+
56
+ Claude Code sends failures on PostToolUseFailure with the error as a
57
+ top-level `error` string (for Bash, first line "Exit code N"); successes
58
+ arrive on PostToolUse. The tool_response checks cover other agents and
59
+ older shapes."""
54
60
  resp = payload.get("tool_response")
55
61
  error = payload.get("error")
56
- text, failed, code = "", bool(error), None
62
+ failure_event = payload.get("hook_event_name") == "PostToolUseFailure"
63
+ text, failed, code = "", bool(error) or failure_event, None
64
+ if isinstance(error, str):
65
+ m = EXIT_PREFIX.match(error)
66
+ if m:
67
+ code = int(m.group(1))
57
68
  if isinstance(resp, dict):
58
69
  code = resp.get("exit_code", resp.get("exitCode", resp.get("returncode")))
59
70
  if isinstance(code, int) and code != 0:
@@ -80,6 +91,72 @@ def failure(payload: dict[str, Any]) -> str | None:
80
91
  return hashlib.sha256(stable.encode()).hexdigest()[:12]
81
92
 
82
93
 
94
+ SESSION_TTL_SECONDS = 7 * 24 * 3600
95
+ PRUNE_EVERY_SECONDS = 3600
96
+
97
+
98
+ def prune_sessions(directory: Path, now: float | None = None) -> None:
99
+ """Delete session state (json, stop, tmp) untouched for SESSION_TTL_SECONDS.
100
+
101
+ Lock files are never deleted: unlinking a lock someone holds would let a
102
+ second hook lock a fresh inode and break mutual exclusion. They are empty,
103
+ so keeping them costs an inode, not space. A session's data is only
104
+ removed while holding its lock without waiting; a busy session is skipped.
105
+ Runs at most once per PRUNE_EVERY_SECONDS and never raises."""
106
+ try:
107
+ now = time.time() if now is None else now
108
+ marker = directory / ".pruned"
109
+ if marker.exists() and now - marker.stat().st_mtime < PRUNE_EVERY_SECONDS:
110
+ return
111
+ directory.mkdir(mode=0o700, parents=True, exist_ok=True)
112
+ marker.touch()
113
+ os.utime(marker, (now, now))
114
+ for f in directory.iterdir():
115
+ if f.name == ".pruned" or f.suffix not in (".json", ".stop", ".tmp"):
116
+ continue
117
+ try:
118
+ if now - f.stat().st_mtime <= SESSION_TTL_SECONDS:
119
+ continue
120
+ _unlink_if_unlocked(f)
121
+ except OSError:
122
+ continue
123
+ except OSError:
124
+ return
125
+
126
+
127
+ TMP_NAME = re.compile(r"^(?P<stem>.+)\.[0-9a-f]{32}\.tmp$")
128
+
129
+
130
+ def _lock_for(f: Path) -> Path:
131
+ """The lock _session_lock takes for this file. State files are
132
+ `<id>.json` / `<id>.stop` and lock `<id>.lock` (with_suffix, so a
133
+ session id may itself contain dots); their temp files are
134
+ `<id>.<uuid>.tmp` (with_suffix replaces .json/.stop), so the stem before
135
+ the uuid is the session id."""
136
+ m = TMP_NAME.match(f.name)
137
+ if m:
138
+ return f.with_name(m.group("stem") + ".lock")
139
+ return f.with_suffix(".lock")
140
+
141
+
142
+ def _unlink_if_unlocked(f: Path) -> None:
143
+ """Remove f only while holding its session lock (non-blocking)."""
144
+ lock = _lock_for(f)
145
+ fd = os.open(lock, os.O_WRONLY | os.O_CREAT, 0o600)
146
+ try:
147
+ try:
148
+ import fcntl
149
+
150
+ fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
151
+ except ImportError:
152
+ pass # no flock (Windows): best effort, as for the hooks themselves
153
+ except OSError:
154
+ return # a hook holds it: this session is in use, keep its data
155
+ f.unlink(missing_ok=True)
156
+ finally:
157
+ os.close(fd)
158
+
159
+
83
160
  def _state_path(session: str | None) -> Path | None:
84
161
  if not session or not SESSION_ID.fullmatch(session):
85
162
  return None
@@ -167,6 +244,9 @@ def handle(payload: dict[str, Any]) -> dict[str, Any] | None:
167
244
  path = _state_path(payload.get("session_id"))
168
245
  if act is None or path is None:
169
246
  return None
247
+ if payload.get("is_interrupt"):
248
+ return None # an abort, not an error the tool reported: not a loop signal
249
+ prune_sessions(path.parent)
170
250
  # Only a hash and the (redacted) action are stored: never file contents.
171
251
  # Hooks for one session can finish together: serialise the
172
252
  # read-append-write so no action is lost.
@@ -187,9 +267,13 @@ def handle(payload: dict[str, Any]) -> dict[str, Any] | None:
187
267
  return None
188
268
  _log(payload.get("session_id"), actions[-1]["act"], count)
189
269
  shown = actions[-1]["act"].split(":", 1)[1][:200]
270
+ event = payload.get("hook_event_name")
271
+ if event not in ("PostToolUse", "PostToolUseFailure"):
272
+ event = "PostToolUseFailure"
190
273
  return {
191
274
  "hookSpecificOutput": {
192
- "hookEventName": "PostToolUse",
275
+ # Answer on the event that fired (PostToolUseFailure for failures).
276
+ "hookEventName": event,
193
277
  "additionalContext": (
194
278
  f"judgetap: `{shown}` has now failed the same way {count} times; "
195
279
  "stop and re-plan (read the error, try a different approach)."
@@ -113,6 +113,9 @@ def _count_path(session: str) -> Path:
113
113
  def _take_block(session: str) -> bool:
114
114
  """Count one block for this session; False once the cap is reached."""
115
115
  path = _count_path(session)
116
+ from judgetap.guard.loop import prune_sessions
117
+
118
+ prune_sessions(path.parent)
116
119
  with _session_lock(path):
117
120
  try:
118
121
  count = int(path.read_text().strip() or 0)
@@ -177,6 +180,11 @@ def handle(payload: dict[str, Any], engine=None) -> dict[str, Any] | None:
177
180
  return None # needs a judge: no rules-only behaviour here
178
181
  transcript = payload.get("transcript_path")
179
182
  task, said = task_and_reply(transcript)
183
+ # Claude Code passes the final reply directly: the transcript is written
184
+ # asynchronously and may not contain it yet at Stop time.
185
+ last = payload.get("last_assistant_message")
186
+ if isinstance(last, str) and last.strip():
187
+ said = last[-ASSISTANT_TAIL_CHARS:]
180
188
  if not task or not said:
181
189
  _log(
182
190
  session,
@@ -16,16 +16,33 @@ def _home(tmp_path, monkeypatch):
16
16
  def call(
17
17
  command, *, fail=True, err="npm ERR! missing script: build", session="s1", code=1
18
18
  ):
19
- payload = {
19
+ """A documented Claude Code payload: failures on PostToolUseFailure with a
20
+ top-level `error` ("Exit code N" first line for Bash), successes on
21
+ PostToolUse with a tool_response."""
22
+ base = {
20
23
  "session_id": session,
24
+ "transcript_path": "/tmp/t.jsonl",
25
+ "cwd": "/tmp",
26
+ "permission_mode": "default",
21
27
  "tool_name": "Bash",
22
- "tool_input": {"command": command},
23
- "tool_response": {
24
- "stdout": "",
25
- "stderr": err if fail else "",
26
- "exit_code": code if fail else 0,
27
- },
28
+ "tool_input": {"command": command, "description": "run"},
29
+ "tool_use_id": "toolu_01ABC",
28
30
  }
31
+ if fail:
32
+ payload = {
33
+ **base,
34
+ "hook_event_name": "PostToolUseFailure",
35
+ "error": f"Exit code {code}\n{err}",
36
+ "is_interrupt": False,
37
+ "duration_ms": 10,
38
+ }
39
+ else:
40
+ payload = {
41
+ **base,
42
+ "hook_event_name": "PostToolUse",
43
+ "tool_response": {"stdout": "ok", "stderr": "", "interrupted": False},
44
+ "duration_ms": 10,
45
+ }
29
46
  out = io.StringIO()
30
47
  assert loop.run(io.StringIO(json.dumps(payload)), out) == 0
31
48
  return json.loads(out.getvalue()) if out.getvalue() else None
@@ -36,7 +53,7 @@ def test_third_identical_failure_adds_a_note():
36
53
  assert call("npm run build") is None
37
54
  out = call("npm run build")
38
55
  ctx = out["hookSpecificOutput"]
39
- assert ctx["hookEventName"] == "PostToolUse"
56
+ assert ctx["hookEventName"] == "PostToolUseFailure"
40
57
  assert (
41
58
  "`npm run build` has now failed the same way 3 times"
42
59
  in ctx["additionalContext"]
@@ -225,3 +242,141 @@ def test_concurrent_hooks_do_not_lose_actions(tmp_path, monkeypatch):
225
242
  for t in threads:
226
243
  t.join()
227
244
  assert len(real_load(tmp_path / "sessions" / "race.json")) == 6
245
+
246
+
247
+ DOC_FAILURE = {
248
+ # Verbatim from code.claude.com/docs/en/hooks#posttoolusefailure-input
249
+ "session_id": "abc123",
250
+ "transcript_path": "/Users/.../.claude/projects/.../00893aaf-19fa-41d2-8238-13269b9b3ca0.jsonl",
251
+ "cwd": "/Users/...",
252
+ "permission_mode": "default",
253
+ "hook_event_name": "PostToolUseFailure",
254
+ "tool_name": "Bash",
255
+ "tool_input": {"command": "npm test", "description": "Run test suite"},
256
+ "tool_use_id": "toolu_01ABC123...",
257
+ "error": "Exit code 1\nError: Cannot find module 'express'",
258
+ "is_interrupt": False,
259
+ "duration_ms": 4187,
260
+ }
261
+
262
+
263
+ def test_documented_failure_payload_triggers_on_the_third_repeat():
264
+ outs = [loop.handle(dict(DOC_FAILURE)) for _ in range(3)]
265
+ assert outs[:2] == [None, None]
266
+ assert outs[2]["hookSpecificOutput"]["hookEventName"] == "PostToolUseFailure"
267
+
268
+
269
+ def test_interrupts_are_not_loop_signals():
270
+ for _ in range(4):
271
+ assert loop.handle({**DOC_FAILURE, "is_interrupt": True}) is None
272
+
273
+
274
+ def test_install_adds_post_tool_use_failure_and_upgrades_old_installs(tmp_path):
275
+ path = tmp_path / "settings.json"
276
+ old = {
277
+ "hooks": {
278
+ "PostToolUse": [
279
+ {
280
+ "matcher": "Bash",
281
+ "hooks": [{"type": "command", "command": POST_COMMAND}],
282
+ }
283
+ ]
284
+ }
285
+ }
286
+ path.write_text(json.dumps(old))
287
+ assert install(path) is True
288
+ hooks = json.loads(path.read_text())["hooks"]
289
+ assert POST_COMMAND in json.dumps(hooks["PostToolUseFailure"])
290
+ assert install(path) is False
291
+ assert uninstall(path) is True
292
+ assert "PostToolUseFailure" not in json.loads(path.read_text()).get("hooks", {})
293
+
294
+
295
+ def test_prune_sessions_removes_old_state_once_an_hour(tmp_path):
296
+ import os
297
+
298
+ d = tmp_path / "sessions"
299
+ d.mkdir()
300
+ old, fresh = d / "a.json", d / "b.json"
301
+ old.write_text("[]")
302
+ fresh.write_text("[]")
303
+ now = 10_000_000.0
304
+ os.utime(old, (now - 8 * 86400, now - 8 * 86400))
305
+ os.utime(fresh, (now - 60, now - 60))
306
+ loop.prune_sessions(d, now=now)
307
+ assert not old.exists() and fresh.exists()
308
+ stale = d / "c.json"
309
+ stale.write_text("[]")
310
+ os.utime(stale, (now - 9 * 86400, now - 9 * 86400))
311
+ loop.prune_sessions(d, now=now + 60) # within the hour: no sweep
312
+ assert stale.exists()
313
+ loop.prune_sessions(d, now=now + 3700)
314
+ assert not stale.exists()
315
+
316
+
317
+ def test_prune_keeps_locks_and_skips_busy_sessions(tmp_path):
318
+ import fcntl
319
+ import os
320
+ import time
321
+
322
+ from judgetap.guard.loop import SESSION_TTL_SECONDS, prune_sessions
323
+
324
+ old = time.time() - SESSION_TTL_SECONDS - 100
325
+ for name in (
326
+ "idle.json",
327
+ "idle.lock",
328
+ "busy.json",
329
+ "busy.lock",
330
+ "idle.stop",
331
+ "x." + "ab" * 16 + ".tmp",
332
+ ):
333
+ p = tmp_path / name
334
+ p.write_text("{}")
335
+ os.utime(p, (old, old))
336
+ fd = os.open(tmp_path / "busy.lock", os.O_WRONLY)
337
+ fcntl.flock(fd, fcntl.LOCK_EX) # another hook is working on "busy"
338
+ try:
339
+ prune_sessions(tmp_path)
340
+ finally:
341
+ os.close(fd)
342
+ left = sorted(p.name for p in tmp_path.iterdir() if p.name != ".pruned")
343
+ # x.lock: created to take the orphan temp file's session lock before deleting it.
344
+ assert left == ["busy.json", "busy.lock", "idle.lock", "x.lock"]
345
+
346
+
347
+ def test_prune_leaves_a_temp_file_whose_session_is_locked(tmp_path):
348
+ import fcntl
349
+ import os
350
+ import time
351
+
352
+ from judgetap.guard.loop import SESSION_TTL_SECONDS, prune_sessions
353
+
354
+ old = time.time() - SESSION_TTL_SECONDS - 100
355
+ tmp = tmp_path / (
356
+ "busy." + "0123abcd" * 4 + ".tmp"
357
+ ) # <id>.<uuid hex>.tmp, as _save names it
358
+ lock = tmp_path / "busy.lock"
359
+ for p in (tmp, lock):
360
+ p.write_text("x")
361
+ os.utime(p, (old, old))
362
+ fd = os.open(lock, os.O_WRONLY)
363
+ fcntl.flock(fd, fcntl.LOCK_EX)
364
+ try:
365
+ prune_sessions(tmp_path)
366
+ finally:
367
+ os.close(fd)
368
+ assert tmp.exists()
369
+ prune_sessions(tmp_path, now=time.time() + 7200) # next sweep, lock free
370
+ assert not tmp.exists()
371
+
372
+
373
+ def test_lock_path_matches_session_lock_for_dotted_ids(tmp_path):
374
+ import uuid
375
+
376
+ from judgetap.guard.loop import _lock_for
377
+
378
+ d = tmp_path
379
+ for data in ("a.b.json", "a.b.stop"):
380
+ assert _lock_for(d / data) == (d / data).with_suffix(".lock") == d / "a.b.lock"
381
+ tmp = (d / data).with_suffix(f".{uuid.uuid4().hex}.tmp") # as _save writes it
382
+ assert _lock_for(tmp) == d / "a.b.lock"
@@ -212,3 +212,40 @@ def test_failed_judge_logs_its_calls_and_lets_the_agent_stop(tmp_path, monkeypat
212
212
  is None
213
213
  )
214
214
  assert load(tmp_path)["summary"]["engines"]["low"]["calls"] == 1
215
+
216
+
217
+ def test_stop_uses_last_assistant_message_when_the_transcript_lags(
218
+ tmp_path, monkeypatch
219
+ ):
220
+ import json
221
+
222
+ from judgetap.guard import stop
223
+ from judgetap.testing import StaticEngine
224
+
225
+ monkeypatch.setenv("JUDGETAP_HOME", str(tmp_path))
226
+ t = tmp_path / "t.jsonl"
227
+ t.write_text(
228
+ json.dumps({"type": "user", "message": {"content": "Fix all 3 failing tests"}})
229
+ + "\n"
230
+ )
231
+ seen = []
232
+
233
+ def judge(q, ctx):
234
+ seen.append(ctx["assistant_last_message"])
235
+ return {"yes": 0.05, "no": 0.95}
236
+
237
+ # The documented Stop input: the final reply comes in last_assistant_message.
238
+ payload = {
239
+ "session_id": "abc123",
240
+ "transcript_path": str(t),
241
+ "cwd": "/tmp",
242
+ "permission_mode": "default",
243
+ "hook_event_name": "Stop",
244
+ "stop_hook_active": False,
245
+ "last_assistant_message": "I fixed one test; two remain, stopping.",
246
+ "background_tasks": [],
247
+ "session_crons": [],
248
+ }
249
+ out = stop.handle(payload, engine=StaticEngine(judge, name="j"))
250
+ assert seen == ["I fixed one test; two remain, stopping."]
251
+ assert out["decision"] == "block"
File without changes
File without changes