judgetap 0.0.2.dev26__tar.gz → 0.0.2.dev27__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/PKG-INFO +2 -2
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/README.md +1 -1
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/docs/SPEC.md +1 -1
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/pyproject.toml +1 -1
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/__init__.py +1 -1
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/guard/install.py +13 -2
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/guard/loop.py +87 -3
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/guard/stop.py +8 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_guard_loop.py +163 -8
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_guard_stop.py +37 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/.github/workflows/ci.yml +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/.github/workflows/demo.yml +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/.github/workflows/release.yml +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/.gitignore +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/.python-version +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/.release-please-manifest.json +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/CONTRIBUTING.md +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/LICENSE +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/docs/demo.tape +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/release-please-config.json +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/_compat.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/api.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/cascade.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/cli.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/dashboard/__init__.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/dashboard/data.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/dashboard/page.html +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/dashboard/server.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/decision_log.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/engine.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/engines/__init__.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/engines/agentjev.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/engines/jev.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/engines/laya.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/engines/llm.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/errors.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/evaluate.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/guard/__init__.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/guard/core.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/guard/hook.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/guard/rules.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/py.typed +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/secrets.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/testing.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/src/judgetap/types.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_api.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_call_accounting.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_calls.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_cascade.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_dashboard.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_decision_log.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_engine_jev.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_engine_llm.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_engine_local.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_evaluate.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_guard_agents.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_guard_core.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_guard_hook.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_guard_rules.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_questions.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/tests/test_secrets.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev27}/uv.lock +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: judgetap
|
|
3
|
-
Version: 0.0.2.
|
|
3
|
+
Version: 0.0.2.dev27
|
|
4
4
|
Summary: Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development.
|
|
5
5
|
Project-URL: Homepage, https://github.com/mergesafe-ai/judgetap
|
|
6
6
|
Author-email: Omer Bar-Ness <omer@zsquared.io>
|
|
@@ -80,7 +80,7 @@ On a CPU-only Linux box, install the CPU PyTorch wheel before `judgetap[laya]` (
|
|
|
80
80
|
|
|
81
81
|
## Guard details
|
|
82
82
|
|
|
83
|
-
**Loop detection (Claude Code).**
|
|
83
|
+
**Loop detection (Claude Code).** `PostToolUse` and `PostToolUseFailure` hooks (`judgetap guard post`, added by `guard install --for claude-code`) notices when the same command or edit fails the same way three times within eight actions and adds a note asking the agent to re-plan. It never blocks, uses no model, and stores only a redacted action and an error hash per session. Cursor and Codex: not yet.
|
|
84
84
|
|
|
85
85
|
- **Agents:** Claude Code (shell, writes and edits), Cursor and Codex (shell only; their hooks don't expose writes and edits).
|
|
86
86
|
- **Engine:** `judgetap guard install` uses one you already have (`$JUDGETAP_ENGINE`, a `TYPESAFE_API_KEY`, or a local AgentJev) and saves it in `~/.judgetap/guard.toml`, because agents often run hooks without your shell's environment.
|
|
@@ -63,7 +63,7 @@ On a CPU-only Linux box, install the CPU PyTorch wheel before `judgetap[laya]` (
|
|
|
63
63
|
|
|
64
64
|
## Guard details
|
|
65
65
|
|
|
66
|
-
**Loop detection (Claude Code).**
|
|
66
|
+
**Loop detection (Claude Code).** `PostToolUse` and `PostToolUseFailure` hooks (`judgetap guard post`, added by `guard install --for claude-code`) notices when the same command or edit fails the same way three times within eight actions and adds a note asking the agent to re-plan. It never blocks, uses no model, and stores only a redacted action and an error hash per session. Cursor and Codex: not yet.
|
|
67
67
|
|
|
68
68
|
- **Agents:** Claude Code (shell, writes and edits), Cursor and Codex (shell only; their hooks don't expose writes and edits).
|
|
69
69
|
- **Engine:** `judgetap guard install` uses one you already have (`$JUDGETAP_ENGINE`, a `TYPESAFE_API_KEY`, or a local AgentJev) and saves it in `~/.judgetap/guard.toml`, because agents often run hooks without your shell's environment.
|
|
@@ -67,7 +67,7 @@ A pre-action hook for coding agents, built on the core.
|
|
|
67
67
|
- **Outcomes**: allow (silent), hold (block with a reason the agent reads and re-plans from), ask (escalate to the user). Holds should be rare; the target is under 5 per 1,000 calls.
|
|
68
68
|
- **Fails safe and visibly**: Claude Code treats a crashing hook as non-blocking, so the guard catches its own errors, applies the rules layer alone, and says so.
|
|
69
69
|
- **Log**: every decision to a local JSONL, so `judgetap guard stats` can report holds and cost. Marking a hold as a false alarm comes with the dashboard (#10).
|
|
70
|
-
- **Loop detection** (Claude Code `PostToolUse
|
|
70
|
+
- **Loop detection** (Claude Code `PostToolUseFailure` for failures, `PostToolUse` for successes that reset a streak; no model): the same action failing with the same error (numbers ignored) 3 times in the last 8 actions adds `additionalContext` telling the agent to re-plan; a success of that action resets the count. Never blocks. Per-session ring buffer of 20 redacted actions and error hashes in `~/.judgetap/sessions/`, 0600. Logged as layer `loop`, outcome `note`.
|
|
71
71
|
|
|
72
72
|
## Decided: the guard's default engine (#8)
|
|
73
73
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "judgetap"
|
|
3
|
-
version = "0.0.2.
|
|
3
|
+
version = "0.0.2.dev27"
|
|
4
4
|
description = "Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "Apache-2.0"
|
|
@@ -89,7 +89,14 @@ def _event(agent: str) -> str:
|
|
|
89
89
|
|
|
90
90
|
def _events(agent: str, with_stop: bool = False) -> list[str]:
|
|
91
91
|
if agent == "claude-code":
|
|
92
|
-
|
|
92
|
+
# Loop detection needs both: failures arrive only on
|
|
93
|
+
# PostToolUseFailure, successes (which end a streak) on PostToolUse.
|
|
94
|
+
return [
|
|
95
|
+
"PreToolUse",
|
|
96
|
+
"PostToolUse",
|
|
97
|
+
"PostToolUseFailure",
|
|
98
|
+
*(["Stop"] if with_stop else []),
|
|
99
|
+
]
|
|
93
100
|
return [_event(agent)]
|
|
94
101
|
|
|
95
102
|
|
|
@@ -98,7 +105,11 @@ def _entry(agent: str, event: str) -> dict:
|
|
|
98
105
|
return {"command": hook_command(agent)}
|
|
99
106
|
if event == "Stop": # Stop takes no matcher
|
|
100
107
|
return {"hooks": [{"type": "command", "command": STOP_COMMAND}]}
|
|
101
|
-
command =
|
|
108
|
+
command = (
|
|
109
|
+
POST_COMMAND
|
|
110
|
+
if event in ("PostToolUse", "PostToolUseFailure")
|
|
111
|
+
else hook_command(agent)
|
|
112
|
+
)
|
|
102
113
|
# Codex's PreToolUse fires for shell only today; the matcher says so.
|
|
103
114
|
matcher = "^(exec_command|shell|Bash)$" if agent == "codex" else MATCHER
|
|
104
115
|
return {"matcher": matcher, "hooks": [{"type": "command", "command": command}]}
|
|
@@ -13,6 +13,7 @@ import json
|
|
|
13
13
|
import os
|
|
14
14
|
import re
|
|
15
15
|
import sys
|
|
16
|
+
import time
|
|
16
17
|
import uuid
|
|
17
18
|
from contextlib import contextmanager
|
|
18
19
|
from datetime import UTC, datetime
|
|
@@ -50,10 +51,20 @@ EXIT_PREFIX = re.compile(r"^\s*exit code[: ]\s*(-?\d+)", re.IGNORECASE)
|
|
|
50
51
|
|
|
51
52
|
|
|
52
53
|
def failure(payload: dict[str, Any]) -> str | None:
|
|
53
|
-
"""A short hash of the error, or None when the call succeeded.
|
|
54
|
+
"""A short hash of the error, or None when the call succeeded.
|
|
55
|
+
|
|
56
|
+
Claude Code sends failures on PostToolUseFailure with the error as a
|
|
57
|
+
top-level `error` string (for Bash, first line "Exit code N"); successes
|
|
58
|
+
arrive on PostToolUse. The tool_response checks cover other agents and
|
|
59
|
+
older shapes."""
|
|
54
60
|
resp = payload.get("tool_response")
|
|
55
61
|
error = payload.get("error")
|
|
56
|
-
|
|
62
|
+
failure_event = payload.get("hook_event_name") == "PostToolUseFailure"
|
|
63
|
+
text, failed, code = "", bool(error) or failure_event, None
|
|
64
|
+
if isinstance(error, str):
|
|
65
|
+
m = EXIT_PREFIX.match(error)
|
|
66
|
+
if m:
|
|
67
|
+
code = int(m.group(1))
|
|
57
68
|
if isinstance(resp, dict):
|
|
58
69
|
code = resp.get("exit_code", resp.get("exitCode", resp.get("returncode")))
|
|
59
70
|
if isinstance(code, int) and code != 0:
|
|
@@ -80,6 +91,72 @@ def failure(payload: dict[str, Any]) -> str | None:
|
|
|
80
91
|
return hashlib.sha256(stable.encode()).hexdigest()[:12]
|
|
81
92
|
|
|
82
93
|
|
|
94
|
+
SESSION_TTL_SECONDS = 7 * 24 * 3600
|
|
95
|
+
PRUNE_EVERY_SECONDS = 3600
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def prune_sessions(directory: Path, now: float | None = None) -> None:
|
|
99
|
+
"""Delete session state (json, stop, tmp) untouched for SESSION_TTL_SECONDS.
|
|
100
|
+
|
|
101
|
+
Lock files are never deleted: unlinking a lock someone holds would let a
|
|
102
|
+
second hook lock a fresh inode and break mutual exclusion. They are empty,
|
|
103
|
+
so keeping them costs an inode, not space. A session's data is only
|
|
104
|
+
removed while holding its lock without waiting; a busy session is skipped.
|
|
105
|
+
Runs at most once per PRUNE_EVERY_SECONDS and never raises."""
|
|
106
|
+
try:
|
|
107
|
+
now = time.time() if now is None else now
|
|
108
|
+
marker = directory / ".pruned"
|
|
109
|
+
if marker.exists() and now - marker.stat().st_mtime < PRUNE_EVERY_SECONDS:
|
|
110
|
+
return
|
|
111
|
+
directory.mkdir(mode=0o700, parents=True, exist_ok=True)
|
|
112
|
+
marker.touch()
|
|
113
|
+
os.utime(marker, (now, now))
|
|
114
|
+
for f in directory.iterdir():
|
|
115
|
+
if f.name == ".pruned" or f.suffix not in (".json", ".stop", ".tmp"):
|
|
116
|
+
continue
|
|
117
|
+
try:
|
|
118
|
+
if now - f.stat().st_mtime <= SESSION_TTL_SECONDS:
|
|
119
|
+
continue
|
|
120
|
+
_unlink_if_unlocked(f)
|
|
121
|
+
except OSError:
|
|
122
|
+
continue
|
|
123
|
+
except OSError:
|
|
124
|
+
return
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
TMP_NAME = re.compile(r"^(?P<stem>.+)\.[0-9a-f]{32}\.tmp$")
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _lock_for(f: Path) -> Path:
|
|
131
|
+
"""The lock _session_lock takes for this file. State files are
|
|
132
|
+
`<id>.json` / `<id>.stop` and lock `<id>.lock` (with_suffix, so a
|
|
133
|
+
session id may itself contain dots); their temp files are
|
|
134
|
+
`<id>.<uuid>.tmp` (with_suffix replaces .json/.stop), so the stem before
|
|
135
|
+
the uuid is the session id."""
|
|
136
|
+
m = TMP_NAME.match(f.name)
|
|
137
|
+
if m:
|
|
138
|
+
return f.with_name(m.group("stem") + ".lock")
|
|
139
|
+
return f.with_suffix(".lock")
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _unlink_if_unlocked(f: Path) -> None:
|
|
143
|
+
"""Remove f only while holding its session lock (non-blocking)."""
|
|
144
|
+
lock = _lock_for(f)
|
|
145
|
+
fd = os.open(lock, os.O_WRONLY | os.O_CREAT, 0o600)
|
|
146
|
+
try:
|
|
147
|
+
try:
|
|
148
|
+
import fcntl
|
|
149
|
+
|
|
150
|
+
fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
151
|
+
except ImportError:
|
|
152
|
+
pass # no flock (Windows): best effort, as for the hooks themselves
|
|
153
|
+
except OSError:
|
|
154
|
+
return # a hook holds it: this session is in use, keep its data
|
|
155
|
+
f.unlink(missing_ok=True)
|
|
156
|
+
finally:
|
|
157
|
+
os.close(fd)
|
|
158
|
+
|
|
159
|
+
|
|
83
160
|
def _state_path(session: str | None) -> Path | None:
|
|
84
161
|
if not session or not SESSION_ID.fullmatch(session):
|
|
85
162
|
return None
|
|
@@ -167,6 +244,9 @@ def handle(payload: dict[str, Any]) -> dict[str, Any] | None:
|
|
|
167
244
|
path = _state_path(payload.get("session_id"))
|
|
168
245
|
if act is None or path is None:
|
|
169
246
|
return None
|
|
247
|
+
if payload.get("is_interrupt"):
|
|
248
|
+
return None # an abort, not an error the tool reported: not a loop signal
|
|
249
|
+
prune_sessions(path.parent)
|
|
170
250
|
# Only a hash and the (redacted) action are stored: never file contents.
|
|
171
251
|
# Hooks for one session can finish together: serialise the
|
|
172
252
|
# read-append-write so no action is lost.
|
|
@@ -187,9 +267,13 @@ def handle(payload: dict[str, Any]) -> dict[str, Any] | None:
|
|
|
187
267
|
return None
|
|
188
268
|
_log(payload.get("session_id"), actions[-1]["act"], count)
|
|
189
269
|
shown = actions[-1]["act"].split(":", 1)[1][:200]
|
|
270
|
+
event = payload.get("hook_event_name")
|
|
271
|
+
if event not in ("PostToolUse", "PostToolUseFailure"):
|
|
272
|
+
event = "PostToolUseFailure"
|
|
190
273
|
return {
|
|
191
274
|
"hookSpecificOutput": {
|
|
192
|
-
|
|
275
|
+
# Answer on the event that fired (PostToolUseFailure for failures).
|
|
276
|
+
"hookEventName": event,
|
|
193
277
|
"additionalContext": (
|
|
194
278
|
f"judgetap: `{shown}` has now failed the same way {count} times; "
|
|
195
279
|
"stop and re-plan (read the error, try a different approach)."
|
|
@@ -113,6 +113,9 @@ def _count_path(session: str) -> Path:
|
|
|
113
113
|
def _take_block(session: str) -> bool:
|
|
114
114
|
"""Count one block for this session; False once the cap is reached."""
|
|
115
115
|
path = _count_path(session)
|
|
116
|
+
from judgetap.guard.loop import prune_sessions
|
|
117
|
+
|
|
118
|
+
prune_sessions(path.parent)
|
|
116
119
|
with _session_lock(path):
|
|
117
120
|
try:
|
|
118
121
|
count = int(path.read_text().strip() or 0)
|
|
@@ -177,6 +180,11 @@ def handle(payload: dict[str, Any], engine=None) -> dict[str, Any] | None:
|
|
|
177
180
|
return None # needs a judge: no rules-only behaviour here
|
|
178
181
|
transcript = payload.get("transcript_path")
|
|
179
182
|
task, said = task_and_reply(transcript)
|
|
183
|
+
# Claude Code passes the final reply directly: the transcript is written
|
|
184
|
+
# asynchronously and may not contain it yet at Stop time.
|
|
185
|
+
last = payload.get("last_assistant_message")
|
|
186
|
+
if isinstance(last, str) and last.strip():
|
|
187
|
+
said = last[-ASSISTANT_TAIL_CHARS:]
|
|
180
188
|
if not task or not said:
|
|
181
189
|
_log(
|
|
182
190
|
session,
|
|
@@ -16,16 +16,33 @@ def _home(tmp_path, monkeypatch):
|
|
|
16
16
|
def call(
|
|
17
17
|
command, *, fail=True, err="npm ERR! missing script: build", session="s1", code=1
|
|
18
18
|
):
|
|
19
|
-
payload
|
|
19
|
+
"""A documented Claude Code payload: failures on PostToolUseFailure with a
|
|
20
|
+
top-level `error` ("Exit code N" first line for Bash), successes on
|
|
21
|
+
PostToolUse with a tool_response."""
|
|
22
|
+
base = {
|
|
20
23
|
"session_id": session,
|
|
24
|
+
"transcript_path": "/tmp/t.jsonl",
|
|
25
|
+
"cwd": "/tmp",
|
|
26
|
+
"permission_mode": "default",
|
|
21
27
|
"tool_name": "Bash",
|
|
22
|
-
"tool_input": {"command": command},
|
|
23
|
-
"
|
|
24
|
-
"stdout": "",
|
|
25
|
-
"stderr": err if fail else "",
|
|
26
|
-
"exit_code": code if fail else 0,
|
|
27
|
-
},
|
|
28
|
+
"tool_input": {"command": command, "description": "run"},
|
|
29
|
+
"tool_use_id": "toolu_01ABC",
|
|
28
30
|
}
|
|
31
|
+
if fail:
|
|
32
|
+
payload = {
|
|
33
|
+
**base,
|
|
34
|
+
"hook_event_name": "PostToolUseFailure",
|
|
35
|
+
"error": f"Exit code {code}\n{err}",
|
|
36
|
+
"is_interrupt": False,
|
|
37
|
+
"duration_ms": 10,
|
|
38
|
+
}
|
|
39
|
+
else:
|
|
40
|
+
payload = {
|
|
41
|
+
**base,
|
|
42
|
+
"hook_event_name": "PostToolUse",
|
|
43
|
+
"tool_response": {"stdout": "ok", "stderr": "", "interrupted": False},
|
|
44
|
+
"duration_ms": 10,
|
|
45
|
+
}
|
|
29
46
|
out = io.StringIO()
|
|
30
47
|
assert loop.run(io.StringIO(json.dumps(payload)), out) == 0
|
|
31
48
|
return json.loads(out.getvalue()) if out.getvalue() else None
|
|
@@ -36,7 +53,7 @@ def test_third_identical_failure_adds_a_note():
|
|
|
36
53
|
assert call("npm run build") is None
|
|
37
54
|
out = call("npm run build")
|
|
38
55
|
ctx = out["hookSpecificOutput"]
|
|
39
|
-
assert ctx["hookEventName"] == "
|
|
56
|
+
assert ctx["hookEventName"] == "PostToolUseFailure"
|
|
40
57
|
assert (
|
|
41
58
|
"`npm run build` has now failed the same way 3 times"
|
|
42
59
|
in ctx["additionalContext"]
|
|
@@ -225,3 +242,141 @@ def test_concurrent_hooks_do_not_lose_actions(tmp_path, monkeypatch):
|
|
|
225
242
|
for t in threads:
|
|
226
243
|
t.join()
|
|
227
244
|
assert len(real_load(tmp_path / "sessions" / "race.json")) == 6
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
DOC_FAILURE = {
|
|
248
|
+
# Verbatim from code.claude.com/docs/en/hooks#posttoolusefailure-input
|
|
249
|
+
"session_id": "abc123",
|
|
250
|
+
"transcript_path": "/Users/.../.claude/projects/.../00893aaf-19fa-41d2-8238-13269b9b3ca0.jsonl",
|
|
251
|
+
"cwd": "/Users/...",
|
|
252
|
+
"permission_mode": "default",
|
|
253
|
+
"hook_event_name": "PostToolUseFailure",
|
|
254
|
+
"tool_name": "Bash",
|
|
255
|
+
"tool_input": {"command": "npm test", "description": "Run test suite"},
|
|
256
|
+
"tool_use_id": "toolu_01ABC123...",
|
|
257
|
+
"error": "Exit code 1\nError: Cannot find module 'express'",
|
|
258
|
+
"is_interrupt": False,
|
|
259
|
+
"duration_ms": 4187,
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def test_documented_failure_payload_triggers_on_the_third_repeat():
|
|
264
|
+
outs = [loop.handle(dict(DOC_FAILURE)) for _ in range(3)]
|
|
265
|
+
assert outs[:2] == [None, None]
|
|
266
|
+
assert outs[2]["hookSpecificOutput"]["hookEventName"] == "PostToolUseFailure"
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def test_interrupts_are_not_loop_signals():
|
|
270
|
+
for _ in range(4):
|
|
271
|
+
assert loop.handle({**DOC_FAILURE, "is_interrupt": True}) is None
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def test_install_adds_post_tool_use_failure_and_upgrades_old_installs(tmp_path):
|
|
275
|
+
path = tmp_path / "settings.json"
|
|
276
|
+
old = {
|
|
277
|
+
"hooks": {
|
|
278
|
+
"PostToolUse": [
|
|
279
|
+
{
|
|
280
|
+
"matcher": "Bash",
|
|
281
|
+
"hooks": [{"type": "command", "command": POST_COMMAND}],
|
|
282
|
+
}
|
|
283
|
+
]
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
path.write_text(json.dumps(old))
|
|
287
|
+
assert install(path) is True
|
|
288
|
+
hooks = json.loads(path.read_text())["hooks"]
|
|
289
|
+
assert POST_COMMAND in json.dumps(hooks["PostToolUseFailure"])
|
|
290
|
+
assert install(path) is False
|
|
291
|
+
assert uninstall(path) is True
|
|
292
|
+
assert "PostToolUseFailure" not in json.loads(path.read_text()).get("hooks", {})
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def test_prune_sessions_removes_old_state_once_an_hour(tmp_path):
|
|
296
|
+
import os
|
|
297
|
+
|
|
298
|
+
d = tmp_path / "sessions"
|
|
299
|
+
d.mkdir()
|
|
300
|
+
old, fresh = d / "a.json", d / "b.json"
|
|
301
|
+
old.write_text("[]")
|
|
302
|
+
fresh.write_text("[]")
|
|
303
|
+
now = 10_000_000.0
|
|
304
|
+
os.utime(old, (now - 8 * 86400, now - 8 * 86400))
|
|
305
|
+
os.utime(fresh, (now - 60, now - 60))
|
|
306
|
+
loop.prune_sessions(d, now=now)
|
|
307
|
+
assert not old.exists() and fresh.exists()
|
|
308
|
+
stale = d / "c.json"
|
|
309
|
+
stale.write_text("[]")
|
|
310
|
+
os.utime(stale, (now - 9 * 86400, now - 9 * 86400))
|
|
311
|
+
loop.prune_sessions(d, now=now + 60) # within the hour: no sweep
|
|
312
|
+
assert stale.exists()
|
|
313
|
+
loop.prune_sessions(d, now=now + 3700)
|
|
314
|
+
assert not stale.exists()
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def test_prune_keeps_locks_and_skips_busy_sessions(tmp_path):
|
|
318
|
+
import fcntl
|
|
319
|
+
import os
|
|
320
|
+
import time
|
|
321
|
+
|
|
322
|
+
from judgetap.guard.loop import SESSION_TTL_SECONDS, prune_sessions
|
|
323
|
+
|
|
324
|
+
old = time.time() - SESSION_TTL_SECONDS - 100
|
|
325
|
+
for name in (
|
|
326
|
+
"idle.json",
|
|
327
|
+
"idle.lock",
|
|
328
|
+
"busy.json",
|
|
329
|
+
"busy.lock",
|
|
330
|
+
"idle.stop",
|
|
331
|
+
"x." + "ab" * 16 + ".tmp",
|
|
332
|
+
):
|
|
333
|
+
p = tmp_path / name
|
|
334
|
+
p.write_text("{}")
|
|
335
|
+
os.utime(p, (old, old))
|
|
336
|
+
fd = os.open(tmp_path / "busy.lock", os.O_WRONLY)
|
|
337
|
+
fcntl.flock(fd, fcntl.LOCK_EX) # another hook is working on "busy"
|
|
338
|
+
try:
|
|
339
|
+
prune_sessions(tmp_path)
|
|
340
|
+
finally:
|
|
341
|
+
os.close(fd)
|
|
342
|
+
left = sorted(p.name for p in tmp_path.iterdir() if p.name != ".pruned")
|
|
343
|
+
# x.lock: created to take the orphan temp file's session lock before deleting it.
|
|
344
|
+
assert left == ["busy.json", "busy.lock", "idle.lock", "x.lock"]
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def test_prune_leaves_a_temp_file_whose_session_is_locked(tmp_path):
|
|
348
|
+
import fcntl
|
|
349
|
+
import os
|
|
350
|
+
import time
|
|
351
|
+
|
|
352
|
+
from judgetap.guard.loop import SESSION_TTL_SECONDS, prune_sessions
|
|
353
|
+
|
|
354
|
+
old = time.time() - SESSION_TTL_SECONDS - 100
|
|
355
|
+
tmp = tmp_path / (
|
|
356
|
+
"busy." + "0123abcd" * 4 + ".tmp"
|
|
357
|
+
) # <id>.<uuid hex>.tmp, as _save names it
|
|
358
|
+
lock = tmp_path / "busy.lock"
|
|
359
|
+
for p in (tmp, lock):
|
|
360
|
+
p.write_text("x")
|
|
361
|
+
os.utime(p, (old, old))
|
|
362
|
+
fd = os.open(lock, os.O_WRONLY)
|
|
363
|
+
fcntl.flock(fd, fcntl.LOCK_EX)
|
|
364
|
+
try:
|
|
365
|
+
prune_sessions(tmp_path)
|
|
366
|
+
finally:
|
|
367
|
+
os.close(fd)
|
|
368
|
+
assert tmp.exists()
|
|
369
|
+
prune_sessions(tmp_path, now=time.time() + 7200) # next sweep, lock free
|
|
370
|
+
assert not tmp.exists()
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
def test_lock_path_matches_session_lock_for_dotted_ids(tmp_path):
|
|
374
|
+
import uuid
|
|
375
|
+
|
|
376
|
+
from judgetap.guard.loop import _lock_for
|
|
377
|
+
|
|
378
|
+
d = tmp_path
|
|
379
|
+
for data in ("a.b.json", "a.b.stop"):
|
|
380
|
+
assert _lock_for(d / data) == (d / data).with_suffix(".lock") == d / "a.b.lock"
|
|
381
|
+
tmp = (d / data).with_suffix(f".{uuid.uuid4().hex}.tmp") # as _save writes it
|
|
382
|
+
assert _lock_for(tmp) == d / "a.b.lock"
|
|
@@ -212,3 +212,40 @@ def test_failed_judge_logs_its_calls_and_lets_the_agent_stop(tmp_path, monkeypat
|
|
|
212
212
|
is None
|
|
213
213
|
)
|
|
214
214
|
assert load(tmp_path)["summary"]["engines"]["low"]["calls"] == 1
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def test_stop_uses_last_assistant_message_when_the_transcript_lags(
|
|
218
|
+
tmp_path, monkeypatch
|
|
219
|
+
):
|
|
220
|
+
import json
|
|
221
|
+
|
|
222
|
+
from judgetap.guard import stop
|
|
223
|
+
from judgetap.testing import StaticEngine
|
|
224
|
+
|
|
225
|
+
monkeypatch.setenv("JUDGETAP_HOME", str(tmp_path))
|
|
226
|
+
t = tmp_path / "t.jsonl"
|
|
227
|
+
t.write_text(
|
|
228
|
+
json.dumps({"type": "user", "message": {"content": "Fix all 3 failing tests"}})
|
|
229
|
+
+ "\n"
|
|
230
|
+
)
|
|
231
|
+
seen = []
|
|
232
|
+
|
|
233
|
+
def judge(q, ctx):
|
|
234
|
+
seen.append(ctx["assistant_last_message"])
|
|
235
|
+
return {"yes": 0.05, "no": 0.95}
|
|
236
|
+
|
|
237
|
+
# The documented Stop input: the final reply comes in last_assistant_message.
|
|
238
|
+
payload = {
|
|
239
|
+
"session_id": "abc123",
|
|
240
|
+
"transcript_path": str(t),
|
|
241
|
+
"cwd": "/tmp",
|
|
242
|
+
"permission_mode": "default",
|
|
243
|
+
"hook_event_name": "Stop",
|
|
244
|
+
"stop_hook_active": False,
|
|
245
|
+
"last_assistant_message": "I fixed one test; two remain, stopping.",
|
|
246
|
+
"background_tasks": [],
|
|
247
|
+
"session_crons": [],
|
|
248
|
+
}
|
|
249
|
+
out = stop.handle(payload, engine=StaticEngine(judge, name="j"))
|
|
250
|
+
assert seen == ["I fixed one test; two remain, stopping."]
|
|
251
|
+
assert out["decision"] == "block"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|