judgetap 0.0.2.dev26__tar.gz → 0.0.2.dev28__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/PKG-INFO +2 -2
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/README.md +1 -1
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/docs/SPEC.md +1 -1
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/pyproject.toml +1 -1
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/__init__.py +1 -1
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/_compat.py +8 -6
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/cli.py +16 -4
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/dashboard/data.py +97 -6
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/dashboard/server.py +9 -1
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/guard/install.py +13 -2
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/guard/loop.py +87 -3
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/guard/stop.py +8 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/secrets.py +3 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_guard_loop.py +163 -8
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_guard_stop.py +37 -0
- judgetap-0.0.2.dev28/tests/test_robustness_68.py +177 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/.github/workflows/ci.yml +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/.github/workflows/demo.yml +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/.github/workflows/release.yml +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/.gitignore +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/.python-version +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/.release-please-manifest.json +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/CONTRIBUTING.md +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/LICENSE +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/docs/demo.tape +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/release-please-config.json +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/api.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/cascade.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/dashboard/__init__.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/dashboard/page.html +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/decision_log.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/engine.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/engines/__init__.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/engines/agentjev.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/engines/jev.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/engines/laya.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/engines/llm.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/errors.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/evaluate.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/guard/__init__.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/guard/core.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/guard/hook.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/guard/rules.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/py.typed +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/testing.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/types.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_api.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_call_accounting.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_calls.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_cascade.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_dashboard.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_decision_log.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_engine_jev.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_engine_llm.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_engine_local.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_evaluate.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_guard_agents.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_guard_core.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_guard_hook.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_guard_rules.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_questions.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_secrets.py +0 -0
- {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/uv.lock +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: judgetap
|
|
3
|
-
Version: 0.0.2.
|
|
3
|
+
Version: 0.0.2.dev28
|
|
4
4
|
Summary: Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development.
|
|
5
5
|
Project-URL: Homepage, https://github.com/mergesafe-ai/judgetap
|
|
6
6
|
Author-email: Omer Bar-Ness <omer@zsquared.io>
|
|
@@ -80,7 +80,7 @@ On a CPU-only Linux box, install the CPU PyTorch wheel before `judgetap[laya]` (
|
|
|
80
80
|
|
|
81
81
|
## Guard details
|
|
82
82
|
|
|
83
|
-
**Loop detection (Claude Code).**
|
|
83
|
+
**Loop detection (Claude Code).** `PostToolUse` and `PostToolUseFailure` hooks (`judgetap guard post`, added by `guard install --for claude-code`) notices when the same command or edit fails the same way three times within eight actions and adds a note asking the agent to re-plan. It never blocks, uses no model, and stores only a redacted action and an error hash per session. Cursor and Codex: not yet.
|
|
84
84
|
|
|
85
85
|
- **Agents:** Claude Code (shell, writes and edits), Cursor and Codex (shell only; their hooks don't expose writes and edits).
|
|
86
86
|
- **Engine:** `judgetap guard install` uses one you already have (`$JUDGETAP_ENGINE`, a `TYPESAFE_API_KEY`, or a local AgentJev) and saves it in `~/.judgetap/guard.toml`, because agents often run hooks without your shell's environment.
|
|
@@ -63,7 +63,7 @@ On a CPU-only Linux box, install the CPU PyTorch wheel before `judgetap[laya]` (
|
|
|
63
63
|
|
|
64
64
|
## Guard details
|
|
65
65
|
|
|
66
|
-
**Loop detection (Claude Code).**
|
|
66
|
+
**Loop detection (Claude Code).** `PostToolUse` and `PostToolUseFailure` hooks (`judgetap guard post`, added by `guard install --for claude-code`) notices when the same command or edit fails the same way three times within eight actions and adds a note asking the agent to re-plan. It never blocks, uses no model, and stores only a redacted action and an error hash per session. Cursor and Codex: not yet.
|
|
67
67
|
|
|
68
68
|
- **Agents:** Claude Code (shell, writes and edits), Cursor and Codex (shell only; their hooks don't expose writes and edits).
|
|
69
69
|
- **Engine:** `judgetap guard install` uses one you already have (`$JUDGETAP_ENGINE`, a `TYPESAFE_API_KEY`, or a local AgentJev) and saves it in `~/.judgetap/guard.toml`, because agents often run hooks without your shell's environment.
|
|
@@ -67,7 +67,7 @@ A pre-action hook for coding agents, built on the core.
|
|
|
67
67
|
- **Outcomes**: allow (silent), hold (block with a reason the agent reads and re-plans from), ask (escalate to the user). Holds should be rare; the target is under 5 per 1,000 calls.
|
|
68
68
|
- **Fails safe and visibly**: Claude Code treats a crashing hook as non-blocking, so the guard catches its own errors, applies the rules layer alone, and says so.
|
|
69
69
|
- **Log**: every decision to a local JSONL, so `judgetap guard stats` can report holds and cost. Marking a hold as a false alarm comes with the dashboard (#10).
|
|
70
|
-
- **Loop detection** (Claude Code `PostToolUse
|
|
70
|
+
- **Loop detection** (Claude Code `PostToolUseFailure` for failures, `PostToolUse` for successes that reset a streak; no model): the same action failing with the same error (numbers ignored) 3 times in the last 8 actions adds `additionalContext` telling the agent to re-plan; a success of that action resets the count. Never blocks. Per-session ring buffer of 20 redacted actions and error hashes in `~/.judgetap/sessions/`, 0600. Logged as layer `loop`, outcome `note`.
|
|
71
71
|
|
|
72
72
|
## Decided: the guard's default engine (#8)
|
|
73
73
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "judgetap"
|
|
3
|
-
version = "0.0.2.
|
|
3
|
+
version = "0.0.2.dev28"
|
|
4
4
|
description = "Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "Apache-2.0"
|
|
@@ -14,8 +14,11 @@ OLD_PREFIX, NEW_PREFIX = "SNAPJUDGE_", "JUDGETAP_"
|
|
|
14
14
|
|
|
15
15
|
|
|
16
16
|
def env(name: str) -> str | None:
|
|
17
|
-
"""$JUDGETAP_<name>, else the old $SNAPJUDGE_<name>.
|
|
18
|
-
|
|
17
|
+
"""$JUDGETAP_<name>, else the old $SNAPJUDGE_<name>. The new name wins
|
|
18
|
+
whenever it is set, even to an empty string (which reads as unset)."""
|
|
19
|
+
if NEW_PREFIX + name in os.environ:
|
|
20
|
+
return os.environ[NEW_PREFIX + name] or None
|
|
21
|
+
return os.environ.get(OLD_PREFIX + name) or None
|
|
19
22
|
|
|
20
23
|
|
|
21
24
|
def default_home() -> Path:
|
|
@@ -26,7 +29,6 @@ def default_home() -> Path:
|
|
|
26
29
|
|
|
27
30
|
def env_source(name: str) -> str | None:
|
|
28
31
|
"""Which variable `env(name)` read: '$JUDGETAP_<name>' or '$SNAPJUDGE_<name>'."""
|
|
29
|
-
|
|
30
|
-
if os.environ
|
|
31
|
-
|
|
32
|
-
return None
|
|
32
|
+
if NEW_PREFIX + name in os.environ: # set, even empty: the old name is ignored
|
|
33
|
+
return f"${NEW_PREFIX}{name}" if os.environ[NEW_PREFIX + name] else None
|
|
34
|
+
return f"${OLD_PREFIX}{name}" if os.environ.get(OLD_PREFIX + name) else None
|
|
@@ -190,7 +190,7 @@ def _dashboard(args) -> int:
|
|
|
190
190
|
from judgetap.guard.hook import home
|
|
191
191
|
|
|
192
192
|
server = serve(home(), args.port)
|
|
193
|
-
url = f"http://127.0.0.1:{
|
|
193
|
+
url = f"http://127.0.0.1:{server.server_address[1]}/" # the bound port (--port 0)
|
|
194
194
|
print(f"judgetap dashboard on {url} (Ctrl+C to stop); reading {home()}")
|
|
195
195
|
if not args.no_browser:
|
|
196
196
|
webbrowser.open(url)
|
|
@@ -238,8 +238,14 @@ def _offer_keychain(stdin=None) -> None:
|
|
|
238
238
|
def _keys_set(args) -> int:
|
|
239
239
|
import getpass
|
|
240
240
|
|
|
241
|
-
from judgetap.secrets import set_key
|
|
241
|
+
from judgetap.secrets import KNOWN_KEYS, set_key
|
|
242
242
|
|
|
243
|
+
if args.name not in KNOWN_KEYS:
|
|
244
|
+
print(
|
|
245
|
+
f"{args.name} isn't read by any engine; known keys: {', '.join(KNOWN_KEYS)}."
|
|
246
|
+
)
|
|
247
|
+
print("The llm engine reads its provider's key from the environment only.")
|
|
248
|
+
return 2
|
|
243
249
|
value = getpass.getpass(f"{args.name}: ") # never echoed, never in argv
|
|
244
250
|
if not value:
|
|
245
251
|
print("Nothing entered; not saved.")
|
|
@@ -254,9 +260,15 @@ def _keys_set(args) -> int:
|
|
|
254
260
|
|
|
255
261
|
|
|
256
262
|
def _keys_status(args) -> int:
|
|
257
|
-
from judgetap.secrets import key_source
|
|
263
|
+
from judgetap.secrets import KNOWN_KEYS, key_source
|
|
258
264
|
|
|
259
|
-
for
|
|
265
|
+
unknown = [n for n in args.names if n not in KNOWN_KEYS]
|
|
266
|
+
if unknown:
|
|
267
|
+
print(
|
|
268
|
+
f"Not read by any engine: {', '.join(unknown)}; known keys: {', '.join(KNOWN_KEYS)}."
|
|
269
|
+
)
|
|
270
|
+
return 2
|
|
271
|
+
for name in args.names or KNOWN_KEYS:
|
|
260
272
|
print(f"{name}: {key_source(name) or 'not set'}")
|
|
261
273
|
return 0
|
|
262
274
|
|
|
@@ -85,8 +85,12 @@ def _load(home: Path, today: date) -> dict[str, Any]:
|
|
|
85
85
|
calls: Counter = Counter() # true totals; the latency deques are capped
|
|
86
86
|
recent: deque = deque(maxlen=MAX_RECENT)
|
|
87
87
|
cost, total, false_holds, library, loops, stops = 0.0, 0, 0, 0, 0, 0
|
|
88
|
-
for n,
|
|
89
|
-
|
|
88
|
+
for n, raw in _iter_jsonl(home / "guard.jsonl"):
|
|
89
|
+
try:
|
|
90
|
+
r = _clean(raw)
|
|
91
|
+
r["id"] = record_id(r, n)
|
|
92
|
+
except Exception: # noqa: BLE001, S112 -- one unreadable record never breaks the page
|
|
93
|
+
continue
|
|
90
94
|
r["source"] = r.get("source") or "guard"
|
|
91
95
|
r["false_alarm"] = r["id"] in false_alarms
|
|
92
96
|
cost += r.get("cost_usd") or 0
|
|
@@ -118,7 +122,8 @@ def _load(home: Path, today: date) -> dict[str, Any]:
|
|
|
118
122
|
"outcomes": dict(outcomes),
|
|
119
123
|
"holds_per_1000": round(1000 * outcomes["hold"] / total, 1) if total else None,
|
|
120
124
|
"false_alarms": false_holds,
|
|
121
|
-
|
|
125
|
+
# Individually finite costs can still sum past float range.
|
|
126
|
+
"cost_usd": round(cost, 6) if math.isfinite(cost) else None,
|
|
122
127
|
"engines": {
|
|
123
128
|
name: {
|
|
124
129
|
"calls": calls[name],
|
|
@@ -132,6 +137,89 @@ def _load(home: Path, today: date) -> dict[str, Any]:
|
|
|
132
137
|
return {"summary": summary, "recent": list(recent)[::-1]}
|
|
133
138
|
|
|
134
139
|
|
|
140
|
+
STR_FIELDS = (
|
|
141
|
+
"id",
|
|
142
|
+
"ts",
|
|
143
|
+
"session",
|
|
144
|
+
"source",
|
|
145
|
+
"tool",
|
|
146
|
+
"subject",
|
|
147
|
+
"outcome",
|
|
148
|
+
"layer",
|
|
149
|
+
"rule",
|
|
150
|
+
"reason",
|
|
151
|
+
"engine",
|
|
152
|
+
"error",
|
|
153
|
+
"call",
|
|
154
|
+
"batch",
|
|
155
|
+
)
|
|
156
|
+
NUM_FIELDS = ("cost_usd", "latency_ms")
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _finite(value: Any) -> float | None:
|
|
160
|
+
"""A finite int/float as float; anything else (str, bool, NaN, inf) is None."""
|
|
161
|
+
if isinstance(value, bool) or not isinstance(value, int | float):
|
|
162
|
+
return None
|
|
163
|
+
try:
|
|
164
|
+
number = float(value) # huge JSON ints overflow here: an invalid field
|
|
165
|
+
except OverflowError:
|
|
166
|
+
return None
|
|
167
|
+
return number if math.isfinite(number) else None
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
MAX_DEPTH = 8 # deeper than any field the page reads
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _scrub(value: Any, depth: int = 0) -> Any:
|
|
174
|
+
"""NaN/Infinity anywhere (even in fields the page doesn't use) become None.
|
|
175
|
+
Nesting past MAX_DEPTH is cut to None, so a pathological unrelated field
|
|
176
|
+
can't exhaust the stack and cost the record."""
|
|
177
|
+
if isinstance(value, float) and not math.isfinite(value):
|
|
178
|
+
return None
|
|
179
|
+
if isinstance(value, dict | list) and depth >= MAX_DEPTH:
|
|
180
|
+
return None
|
|
181
|
+
if isinstance(value, dict):
|
|
182
|
+
return {k: _scrub(v, depth + 1) for k, v in value.items()}
|
|
183
|
+
if isinstance(value, list):
|
|
184
|
+
return [_scrub(v, depth + 1) for v in value]
|
|
185
|
+
return value
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _clean(r: dict[str, Any]) -> dict[str, Any]:
|
|
189
|
+
"""The log is ours but may be edited, truncated or written by an older
|
|
190
|
+
version: coerce every field the page uses to its expected type, so no
|
|
191
|
+
value can crash the summary or make the JSON invalid (NaN)."""
|
|
192
|
+
out = _scrub(dict(r))
|
|
193
|
+
for key in STR_FIELDS:
|
|
194
|
+
if key in out and not isinstance(out[key], str):
|
|
195
|
+
out[key] = None
|
|
196
|
+
for key in NUM_FIELDS:
|
|
197
|
+
if key in out:
|
|
198
|
+
out[key] = _finite(out[key])
|
|
199
|
+
if not isinstance(out.get("p"), dict):
|
|
200
|
+
out["p"] = {}
|
|
201
|
+
else:
|
|
202
|
+
out["p"] = {
|
|
203
|
+
k: v
|
|
204
|
+
for k, v in out["p"].items()
|
|
205
|
+
if isinstance(k, str) and _finite(v) is not None
|
|
206
|
+
}
|
|
207
|
+
if "calls" in out and not isinstance(out["calls"], list):
|
|
208
|
+
# Not a calls list at all: treat the record as having none, so the
|
|
209
|
+
# legacy fallback still counts its engine.
|
|
210
|
+
del out["calls"]
|
|
211
|
+
if "calls" in out:
|
|
212
|
+
# Calls are echoed back in "recent" too: keep only well-formed ones
|
|
213
|
+
# with a finite (or absent) latency.
|
|
214
|
+
calls = out["calls"]
|
|
215
|
+
out["calls"] = [
|
|
216
|
+
{**c, "latency_ms": _finite(c.get("latency_ms"))}
|
|
217
|
+
for c in calls
|
|
218
|
+
if isinstance(c, dict) and isinstance(c.get("engine"), str)
|
|
219
|
+
]
|
|
220
|
+
return out
|
|
221
|
+
|
|
222
|
+
|
|
135
223
|
def _count_calls(r: dict[str, Any], calls: Counter, by_engine) -> None:
|
|
136
224
|
"""Engine metrics come from the calls each record reports, each with its
|
|
137
225
|
own latency: nothing is inferred. A batch writes its calls on one record
|
|
@@ -143,14 +231,17 @@ def _count_calls(r: dict[str, Any], calls: Counter, by_engine) -> None:
|
|
|
143
231
|
for c in reported:
|
|
144
232
|
if isinstance(c, dict) and isinstance(c.get("engine"), str):
|
|
145
233
|
calls[c["engine"]] += 1
|
|
146
|
-
|
|
147
|
-
|
|
234
|
+
latency = _finite(c.get("latency_ms"))
|
|
235
|
+
if latency is not None:
|
|
236
|
+
by_engine[c["engine"]].append(latency)
|
|
148
237
|
return
|
|
149
238
|
if not r.get("engine") or r.get("error"):
|
|
150
239
|
return
|
|
151
240
|
if r.get("layer") == "judge":
|
|
152
241
|
calls[r["engine"]] += 1
|
|
153
|
-
|
|
242
|
+
latency = _finite(r.get("latency_ms"))
|
|
243
|
+
if latency is not None: # no sample for a missing or invalid latency
|
|
244
|
+
by_engine[r["engine"]].append(latency)
|
|
154
245
|
elif r.get("layer") == "stop" and r.get("outcome") in ("allow", "block"):
|
|
155
246
|
# A Stop check from before calls were logged: it asked its engine
|
|
156
247
|
# once, but recorded no latency, so only the call is counted.
|
|
@@ -44,7 +44,10 @@ def make_handler(home: Path, token: str, port: int) -> type[BaseHTTPRequestHandl
|
|
|
44
44
|
self.wfile.write(body)
|
|
45
45
|
|
|
46
46
|
def _json(self, status: int, obj: object) -> None:
|
|
47
|
-
|
|
47
|
+
# allow_nan=False: bare NaN isn't JSON and would break the page.
|
|
48
|
+
self._send(
|
|
49
|
+
status, json.dumps(obj, allow_nan=False).encode(), "application/json"
|
|
50
|
+
)
|
|
48
51
|
|
|
49
52
|
def do_GET(self) -> None:
|
|
50
53
|
if not self._host_ok():
|
|
@@ -94,4 +97,9 @@ def make_handler(home: Path, token: str, port: int) -> type[BaseHTTPRequestHandl
|
|
|
94
97
|
def serve(home: Path, port: int = 8765) -> ThreadingHTTPServer:
|
|
95
98
|
token = secrets.token_urlsafe(24)
|
|
96
99
|
server = ThreadingHTTPServer(("127.0.0.1", port), make_handler(home, token, port))
|
|
100
|
+
# Port 0 means "any free port": the Host allow-list must use the port
|
|
101
|
+
# the OS actually bound, or every request is refused.
|
|
102
|
+
bound = server.server_address[1]
|
|
103
|
+
if bound != port:
|
|
104
|
+
server.RequestHandlerClass = make_handler(home, token, bound)
|
|
97
105
|
return server
|
|
@@ -89,7 +89,14 @@ def _event(agent: str) -> str:
|
|
|
89
89
|
|
|
90
90
|
def _events(agent: str, with_stop: bool = False) -> list[str]:
|
|
91
91
|
if agent == "claude-code":
|
|
92
|
-
|
|
92
|
+
# Loop detection needs both: failures arrive only on
|
|
93
|
+
# PostToolUseFailure, successes (which end a streak) on PostToolUse.
|
|
94
|
+
return [
|
|
95
|
+
"PreToolUse",
|
|
96
|
+
"PostToolUse",
|
|
97
|
+
"PostToolUseFailure",
|
|
98
|
+
*(["Stop"] if with_stop else []),
|
|
99
|
+
]
|
|
93
100
|
return [_event(agent)]
|
|
94
101
|
|
|
95
102
|
|
|
@@ -98,7 +105,11 @@ def _entry(agent: str, event: str) -> dict:
|
|
|
98
105
|
return {"command": hook_command(agent)}
|
|
99
106
|
if event == "Stop": # Stop takes no matcher
|
|
100
107
|
return {"hooks": [{"type": "command", "command": STOP_COMMAND}]}
|
|
101
|
-
command =
|
|
108
|
+
command = (
|
|
109
|
+
POST_COMMAND
|
|
110
|
+
if event in ("PostToolUse", "PostToolUseFailure")
|
|
111
|
+
else hook_command(agent)
|
|
112
|
+
)
|
|
102
113
|
# Codex's PreToolUse fires for shell only today; the matcher says so.
|
|
103
114
|
matcher = "^(exec_command|shell|Bash)$" if agent == "codex" else MATCHER
|
|
104
115
|
return {"matcher": matcher, "hooks": [{"type": "command", "command": command}]}
|
|
@@ -13,6 +13,7 @@ import json
|
|
|
13
13
|
import os
|
|
14
14
|
import re
|
|
15
15
|
import sys
|
|
16
|
+
import time
|
|
16
17
|
import uuid
|
|
17
18
|
from contextlib import contextmanager
|
|
18
19
|
from datetime import UTC, datetime
|
|
@@ -50,10 +51,20 @@ EXIT_PREFIX = re.compile(r"^\s*exit code[: ]\s*(-?\d+)", re.IGNORECASE)
|
|
|
50
51
|
|
|
51
52
|
|
|
52
53
|
def failure(payload: dict[str, Any]) -> str | None:
|
|
53
|
-
"""A short hash of the error, or None when the call succeeded.
|
|
54
|
+
"""A short hash of the error, or None when the call succeeded.
|
|
55
|
+
|
|
56
|
+
Claude Code sends failures on PostToolUseFailure with the error as a
|
|
57
|
+
top-level `error` string (for Bash, first line "Exit code N"); successes
|
|
58
|
+
arrive on PostToolUse. The tool_response checks cover other agents and
|
|
59
|
+
older shapes."""
|
|
54
60
|
resp = payload.get("tool_response")
|
|
55
61
|
error = payload.get("error")
|
|
56
|
-
|
|
62
|
+
failure_event = payload.get("hook_event_name") == "PostToolUseFailure"
|
|
63
|
+
text, failed, code = "", bool(error) or failure_event, None
|
|
64
|
+
if isinstance(error, str):
|
|
65
|
+
m = EXIT_PREFIX.match(error)
|
|
66
|
+
if m:
|
|
67
|
+
code = int(m.group(1))
|
|
57
68
|
if isinstance(resp, dict):
|
|
58
69
|
code = resp.get("exit_code", resp.get("exitCode", resp.get("returncode")))
|
|
59
70
|
if isinstance(code, int) and code != 0:
|
|
@@ -80,6 +91,72 @@ def failure(payload: dict[str, Any]) -> str | None:
|
|
|
80
91
|
return hashlib.sha256(stable.encode()).hexdigest()[:12]
|
|
81
92
|
|
|
82
93
|
|
|
94
|
+
SESSION_TTL_SECONDS = 7 * 24 * 3600
|
|
95
|
+
PRUNE_EVERY_SECONDS = 3600
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def prune_sessions(directory: Path, now: float | None = None) -> None:
|
|
99
|
+
"""Delete session state (json, stop, tmp) untouched for SESSION_TTL_SECONDS.
|
|
100
|
+
|
|
101
|
+
Lock files are never deleted: unlinking a lock someone holds would let a
|
|
102
|
+
second hook lock a fresh inode and break mutual exclusion. They are empty,
|
|
103
|
+
so keeping them costs an inode, not space. A session's data is only
|
|
104
|
+
removed while holding its lock without waiting; a busy session is skipped.
|
|
105
|
+
Runs at most once per PRUNE_EVERY_SECONDS and never raises."""
|
|
106
|
+
try:
|
|
107
|
+
now = time.time() if now is None else now
|
|
108
|
+
marker = directory / ".pruned"
|
|
109
|
+
if marker.exists() and now - marker.stat().st_mtime < PRUNE_EVERY_SECONDS:
|
|
110
|
+
return
|
|
111
|
+
directory.mkdir(mode=0o700, parents=True, exist_ok=True)
|
|
112
|
+
marker.touch()
|
|
113
|
+
os.utime(marker, (now, now))
|
|
114
|
+
for f in directory.iterdir():
|
|
115
|
+
if f.name == ".pruned" or f.suffix not in (".json", ".stop", ".tmp"):
|
|
116
|
+
continue
|
|
117
|
+
try:
|
|
118
|
+
if now - f.stat().st_mtime <= SESSION_TTL_SECONDS:
|
|
119
|
+
continue
|
|
120
|
+
_unlink_if_unlocked(f)
|
|
121
|
+
except OSError:
|
|
122
|
+
continue
|
|
123
|
+
except OSError:
|
|
124
|
+
return
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
TMP_NAME = re.compile(r"^(?P<stem>.+)\.[0-9a-f]{32}\.tmp$")
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _lock_for(f: Path) -> Path:
|
|
131
|
+
"""The lock _session_lock takes for this file. State files are
|
|
132
|
+
`<id>.json` / `<id>.stop` and lock `<id>.lock` (with_suffix, so a
|
|
133
|
+
session id may itself contain dots); their temp files are
|
|
134
|
+
`<id>.<uuid>.tmp` (with_suffix replaces .json/.stop), so the stem before
|
|
135
|
+
the uuid is the session id."""
|
|
136
|
+
m = TMP_NAME.match(f.name)
|
|
137
|
+
if m:
|
|
138
|
+
return f.with_name(m.group("stem") + ".lock")
|
|
139
|
+
return f.with_suffix(".lock")
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _unlink_if_unlocked(f: Path) -> None:
|
|
143
|
+
"""Remove f only while holding its session lock (non-blocking)."""
|
|
144
|
+
lock = _lock_for(f)
|
|
145
|
+
fd = os.open(lock, os.O_WRONLY | os.O_CREAT, 0o600)
|
|
146
|
+
try:
|
|
147
|
+
try:
|
|
148
|
+
import fcntl
|
|
149
|
+
|
|
150
|
+
fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
151
|
+
except ImportError:
|
|
152
|
+
pass # no flock (Windows): best effort, as for the hooks themselves
|
|
153
|
+
except OSError:
|
|
154
|
+
return # a hook holds it: this session is in use, keep its data
|
|
155
|
+
f.unlink(missing_ok=True)
|
|
156
|
+
finally:
|
|
157
|
+
os.close(fd)
|
|
158
|
+
|
|
159
|
+
|
|
83
160
|
def _state_path(session: str | None) -> Path | None:
|
|
84
161
|
if not session or not SESSION_ID.fullmatch(session):
|
|
85
162
|
return None
|
|
@@ -167,6 +244,9 @@ def handle(payload: dict[str, Any]) -> dict[str, Any] | None:
|
|
|
167
244
|
path = _state_path(payload.get("session_id"))
|
|
168
245
|
if act is None or path is None:
|
|
169
246
|
return None
|
|
247
|
+
if payload.get("is_interrupt"):
|
|
248
|
+
return None # an abort, not an error the tool reported: not a loop signal
|
|
249
|
+
prune_sessions(path.parent)
|
|
170
250
|
# Only a hash and the (redacted) action are stored: never file contents.
|
|
171
251
|
# Hooks for one session can finish together: serialise the
|
|
172
252
|
# read-append-write so no action is lost.
|
|
@@ -187,9 +267,13 @@ def handle(payload: dict[str, Any]) -> dict[str, Any] | None:
|
|
|
187
267
|
return None
|
|
188
268
|
_log(payload.get("session_id"), actions[-1]["act"], count)
|
|
189
269
|
shown = actions[-1]["act"].split(":", 1)[1][:200]
|
|
270
|
+
event = payload.get("hook_event_name")
|
|
271
|
+
if event not in ("PostToolUse", "PostToolUseFailure"):
|
|
272
|
+
event = "PostToolUseFailure"
|
|
190
273
|
return {
|
|
191
274
|
"hookSpecificOutput": {
|
|
192
|
-
|
|
275
|
+
# Answer on the event that fired (PostToolUseFailure for failures).
|
|
276
|
+
"hookEventName": event,
|
|
193
277
|
"additionalContext": (
|
|
194
278
|
f"judgetap: `{shown}` has now failed the same way {count} times; "
|
|
195
279
|
"stop and re-plan (read the error, try a different approach)."
|
|
@@ -113,6 +113,9 @@ def _count_path(session: str) -> Path:
|
|
|
113
113
|
def _take_block(session: str) -> bool:
|
|
114
114
|
"""Count one block for this session; False once the cap is reached."""
|
|
115
115
|
path = _count_path(session)
|
|
116
|
+
from judgetap.guard.loop import prune_sessions
|
|
117
|
+
|
|
118
|
+
prune_sessions(path.parent)
|
|
116
119
|
with _session_lock(path):
|
|
117
120
|
try:
|
|
118
121
|
count = int(path.read_text().strip() or 0)
|
|
@@ -177,6 +180,11 @@ def handle(payload: dict[str, Any], engine=None) -> dict[str, Any] | None:
|
|
|
177
180
|
return None # needs a judge: no rules-only behaviour here
|
|
178
181
|
transcript = payload.get("transcript_path")
|
|
179
182
|
task, said = task_and_reply(transcript)
|
|
183
|
+
# Claude Code passes the final reply directly: the transcript is written
|
|
184
|
+
# asynchronously and may not contain it yet at Stop time.
|
|
185
|
+
last = payload.get("last_assistant_message")
|
|
186
|
+
if isinstance(last, str) and last.strip():
|
|
187
|
+
said = last[-ASSISTANT_TAIL_CHARS:]
|
|
180
188
|
if not task or not said:
|
|
181
189
|
_log(
|
|
182
190
|
session,
|
|
@@ -10,6 +10,9 @@ from __future__ import annotations
|
|
|
10
10
|
|
|
11
11
|
import os
|
|
12
12
|
|
|
13
|
+
# Keys an engine reads through get_key. Saving any other name to the keychain
|
|
14
|
+
# would store something nothing reads.
|
|
15
|
+
KNOWN_KEYS = ("TYPESAFE_API_KEY",)
|
|
13
16
|
SERVICE = "judgetap"
|
|
14
17
|
OLD_SERVICE = "snapjudge" # keys saved before the rename (#38)
|
|
15
18
|
|
|
@@ -16,16 +16,33 @@ def _home(tmp_path, monkeypatch):
|
|
|
16
16
|
def call(
|
|
17
17
|
command, *, fail=True, err="npm ERR! missing script: build", session="s1", code=1
|
|
18
18
|
):
|
|
19
|
-
payload
|
|
19
|
+
"""A documented Claude Code payload: failures on PostToolUseFailure with a
|
|
20
|
+
top-level `error` ("Exit code N" first line for Bash), successes on
|
|
21
|
+
PostToolUse with a tool_response."""
|
|
22
|
+
base = {
|
|
20
23
|
"session_id": session,
|
|
24
|
+
"transcript_path": "/tmp/t.jsonl",
|
|
25
|
+
"cwd": "/tmp",
|
|
26
|
+
"permission_mode": "default",
|
|
21
27
|
"tool_name": "Bash",
|
|
22
|
-
"tool_input": {"command": command},
|
|
23
|
-
"
|
|
24
|
-
"stdout": "",
|
|
25
|
-
"stderr": err if fail else "",
|
|
26
|
-
"exit_code": code if fail else 0,
|
|
27
|
-
},
|
|
28
|
+
"tool_input": {"command": command, "description": "run"},
|
|
29
|
+
"tool_use_id": "toolu_01ABC",
|
|
28
30
|
}
|
|
31
|
+
if fail:
|
|
32
|
+
payload = {
|
|
33
|
+
**base,
|
|
34
|
+
"hook_event_name": "PostToolUseFailure",
|
|
35
|
+
"error": f"Exit code {code}\n{err}",
|
|
36
|
+
"is_interrupt": False,
|
|
37
|
+
"duration_ms": 10,
|
|
38
|
+
}
|
|
39
|
+
else:
|
|
40
|
+
payload = {
|
|
41
|
+
**base,
|
|
42
|
+
"hook_event_name": "PostToolUse",
|
|
43
|
+
"tool_response": {"stdout": "ok", "stderr": "", "interrupted": False},
|
|
44
|
+
"duration_ms": 10,
|
|
45
|
+
}
|
|
29
46
|
out = io.StringIO()
|
|
30
47
|
assert loop.run(io.StringIO(json.dumps(payload)), out) == 0
|
|
31
48
|
return json.loads(out.getvalue()) if out.getvalue() else None
|
|
@@ -36,7 +53,7 @@ def test_third_identical_failure_adds_a_note():
|
|
|
36
53
|
assert call("npm run build") is None
|
|
37
54
|
out = call("npm run build")
|
|
38
55
|
ctx = out["hookSpecificOutput"]
|
|
39
|
-
assert ctx["hookEventName"] == "
|
|
56
|
+
assert ctx["hookEventName"] == "PostToolUseFailure"
|
|
40
57
|
assert (
|
|
41
58
|
"`npm run build` has now failed the same way 3 times"
|
|
42
59
|
in ctx["additionalContext"]
|
|
@@ -225,3 +242,141 @@ def test_concurrent_hooks_do_not_lose_actions(tmp_path, monkeypatch):
|
|
|
225
242
|
for t in threads:
|
|
226
243
|
t.join()
|
|
227
244
|
assert len(real_load(tmp_path / "sessions" / "race.json")) == 6
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
DOC_FAILURE = {
|
|
248
|
+
# Verbatim from code.claude.com/docs/en/hooks#posttoolusefailure-input
|
|
249
|
+
"session_id": "abc123",
|
|
250
|
+
"transcript_path": "/Users/.../.claude/projects/.../00893aaf-19fa-41d2-8238-13269b9b3ca0.jsonl",
|
|
251
|
+
"cwd": "/Users/...",
|
|
252
|
+
"permission_mode": "default",
|
|
253
|
+
"hook_event_name": "PostToolUseFailure",
|
|
254
|
+
"tool_name": "Bash",
|
|
255
|
+
"tool_input": {"command": "npm test", "description": "Run test suite"},
|
|
256
|
+
"tool_use_id": "toolu_01ABC123...",
|
|
257
|
+
"error": "Exit code 1\nError: Cannot find module 'express'",
|
|
258
|
+
"is_interrupt": False,
|
|
259
|
+
"duration_ms": 4187,
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def test_documented_failure_payload_triggers_on_the_third_repeat():
|
|
264
|
+
outs = [loop.handle(dict(DOC_FAILURE)) for _ in range(3)]
|
|
265
|
+
assert outs[:2] == [None, None]
|
|
266
|
+
assert outs[2]["hookSpecificOutput"]["hookEventName"] == "PostToolUseFailure"
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def test_interrupts_are_not_loop_signals():
|
|
270
|
+
for _ in range(4):
|
|
271
|
+
assert loop.handle({**DOC_FAILURE, "is_interrupt": True}) is None
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def test_install_adds_post_tool_use_failure_and_upgrades_old_installs(tmp_path):
|
|
275
|
+
path = tmp_path / "settings.json"
|
|
276
|
+
old = {
|
|
277
|
+
"hooks": {
|
|
278
|
+
"PostToolUse": [
|
|
279
|
+
{
|
|
280
|
+
"matcher": "Bash",
|
|
281
|
+
"hooks": [{"type": "command", "command": POST_COMMAND}],
|
|
282
|
+
}
|
|
283
|
+
]
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
path.write_text(json.dumps(old))
|
|
287
|
+
assert install(path) is True
|
|
288
|
+
hooks = json.loads(path.read_text())["hooks"]
|
|
289
|
+
assert POST_COMMAND in json.dumps(hooks["PostToolUseFailure"])
|
|
290
|
+
assert install(path) is False
|
|
291
|
+
assert uninstall(path) is True
|
|
292
|
+
assert "PostToolUseFailure" not in json.loads(path.read_text()).get("hooks", {})
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def test_prune_sessions_removes_old_state_once_an_hour(tmp_path):
|
|
296
|
+
import os
|
|
297
|
+
|
|
298
|
+
d = tmp_path / "sessions"
|
|
299
|
+
d.mkdir()
|
|
300
|
+
old, fresh = d / "a.json", d / "b.json"
|
|
301
|
+
old.write_text("[]")
|
|
302
|
+
fresh.write_text("[]")
|
|
303
|
+
now = 10_000_000.0
|
|
304
|
+
os.utime(old, (now - 8 * 86400, now - 8 * 86400))
|
|
305
|
+
os.utime(fresh, (now - 60, now - 60))
|
|
306
|
+
loop.prune_sessions(d, now=now)
|
|
307
|
+
assert not old.exists() and fresh.exists()
|
|
308
|
+
stale = d / "c.json"
|
|
309
|
+
stale.write_text("[]")
|
|
310
|
+
os.utime(stale, (now - 9 * 86400, now - 9 * 86400))
|
|
311
|
+
loop.prune_sessions(d, now=now + 60) # within the hour: no sweep
|
|
312
|
+
assert stale.exists()
|
|
313
|
+
loop.prune_sessions(d, now=now + 3700)
|
|
314
|
+
assert not stale.exists()
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def test_prune_keeps_locks_and_skips_busy_sessions(tmp_path):
|
|
318
|
+
import fcntl
|
|
319
|
+
import os
|
|
320
|
+
import time
|
|
321
|
+
|
|
322
|
+
from judgetap.guard.loop import SESSION_TTL_SECONDS, prune_sessions
|
|
323
|
+
|
|
324
|
+
old = time.time() - SESSION_TTL_SECONDS - 100
|
|
325
|
+
for name in (
|
|
326
|
+
"idle.json",
|
|
327
|
+
"idle.lock",
|
|
328
|
+
"busy.json",
|
|
329
|
+
"busy.lock",
|
|
330
|
+
"idle.stop",
|
|
331
|
+
"x." + "ab" * 16 + ".tmp",
|
|
332
|
+
):
|
|
333
|
+
p = tmp_path / name
|
|
334
|
+
p.write_text("{}")
|
|
335
|
+
os.utime(p, (old, old))
|
|
336
|
+
fd = os.open(tmp_path / "busy.lock", os.O_WRONLY)
|
|
337
|
+
fcntl.flock(fd, fcntl.LOCK_EX) # another hook is working on "busy"
|
|
338
|
+
try:
|
|
339
|
+
prune_sessions(tmp_path)
|
|
340
|
+
finally:
|
|
341
|
+
os.close(fd)
|
|
342
|
+
left = sorted(p.name for p in tmp_path.iterdir() if p.name != ".pruned")
|
|
343
|
+
# x.lock: created to take the orphan temp file's session lock before deleting it.
|
|
344
|
+
assert left == ["busy.json", "busy.lock", "idle.lock", "x.lock"]
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def test_prune_leaves_a_temp_file_whose_session_is_locked(tmp_path):
|
|
348
|
+
import fcntl
|
|
349
|
+
import os
|
|
350
|
+
import time
|
|
351
|
+
|
|
352
|
+
from judgetap.guard.loop import SESSION_TTL_SECONDS, prune_sessions
|
|
353
|
+
|
|
354
|
+
old = time.time() - SESSION_TTL_SECONDS - 100
|
|
355
|
+
tmp = tmp_path / (
|
|
356
|
+
"busy." + "0123abcd" * 4 + ".tmp"
|
|
357
|
+
) # <id>.<uuid hex>.tmp, as _save names it
|
|
358
|
+
lock = tmp_path / "busy.lock"
|
|
359
|
+
for p in (tmp, lock):
|
|
360
|
+
p.write_text("x")
|
|
361
|
+
os.utime(p, (old, old))
|
|
362
|
+
fd = os.open(lock, os.O_WRONLY)
|
|
363
|
+
fcntl.flock(fd, fcntl.LOCK_EX)
|
|
364
|
+
try:
|
|
365
|
+
prune_sessions(tmp_path)
|
|
366
|
+
finally:
|
|
367
|
+
os.close(fd)
|
|
368
|
+
assert tmp.exists()
|
|
369
|
+
prune_sessions(tmp_path, now=time.time() + 7200) # next sweep, lock free
|
|
370
|
+
assert not tmp.exists()
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
def test_lock_path_matches_session_lock_for_dotted_ids(tmp_path):
|
|
374
|
+
import uuid
|
|
375
|
+
|
|
376
|
+
from judgetap.guard.loop import _lock_for
|
|
377
|
+
|
|
378
|
+
d = tmp_path
|
|
379
|
+
for data in ("a.b.json", "a.b.stop"):
|
|
380
|
+
assert _lock_for(d / data) == (d / data).with_suffix(".lock") == d / "a.b.lock"
|
|
381
|
+
tmp = (d / data).with_suffix(f".{uuid.uuid4().hex}.tmp") # as _save writes it
|
|
382
|
+
assert _lock_for(tmp) == d / "a.b.lock"
|
|
@@ -212,3 +212,40 @@ def test_failed_judge_logs_its_calls_and_lets_the_agent_stop(tmp_path, monkeypat
|
|
|
212
212
|
is None
|
|
213
213
|
)
|
|
214
214
|
assert load(tmp_path)["summary"]["engines"]["low"]["calls"] == 1
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def test_stop_uses_last_assistant_message_when_the_transcript_lags(
|
|
218
|
+
tmp_path, monkeypatch
|
|
219
|
+
):
|
|
220
|
+
import json
|
|
221
|
+
|
|
222
|
+
from judgetap.guard import stop
|
|
223
|
+
from judgetap.testing import StaticEngine
|
|
224
|
+
|
|
225
|
+
monkeypatch.setenv("JUDGETAP_HOME", str(tmp_path))
|
|
226
|
+
t = tmp_path / "t.jsonl"
|
|
227
|
+
t.write_text(
|
|
228
|
+
json.dumps({"type": "user", "message": {"content": "Fix all 3 failing tests"}})
|
|
229
|
+
+ "\n"
|
|
230
|
+
)
|
|
231
|
+
seen = []
|
|
232
|
+
|
|
233
|
+
def judge(q, ctx):
|
|
234
|
+
seen.append(ctx["assistant_last_message"])
|
|
235
|
+
return {"yes": 0.05, "no": 0.95}
|
|
236
|
+
|
|
237
|
+
# The documented Stop input: the final reply comes in last_assistant_message.
|
|
238
|
+
payload = {
|
|
239
|
+
"session_id": "abc123",
|
|
240
|
+
"transcript_path": str(t),
|
|
241
|
+
"cwd": "/tmp",
|
|
242
|
+
"permission_mode": "default",
|
|
243
|
+
"hook_event_name": "Stop",
|
|
244
|
+
"stop_hook_active": False,
|
|
245
|
+
"last_assistant_message": "I fixed one test; two remain, stopping.",
|
|
246
|
+
"background_tasks": [],
|
|
247
|
+
"session_crons": [],
|
|
248
|
+
}
|
|
249
|
+
out = stop.handle(payload, engine=StaticEngine(judge, name="j"))
|
|
250
|
+
assert seen == ["I fixed one test; two remain, stopping."]
|
|
251
|
+
assert out["decision"] == "block"
|
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import math
|
|
3
|
+
import threading
|
|
4
|
+
import urllib.request
|
|
5
|
+
|
|
6
|
+
import pytest
|
|
7
|
+
|
|
8
|
+
from judgetap.dashboard.data import load
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def write(tmp_path, rows):
|
|
12
|
+
(tmp_path / "guard.jsonl").write_text("\n".join(rows) + "\n")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@pytest.mark.parametrize(
|
|
16
|
+
"line",
|
|
17
|
+
[
|
|
18
|
+
'{"outcome":"allow","cost_usd":"0.1"}',
|
|
19
|
+
'{"outcome":["hold"]}',
|
|
20
|
+
'{"outcome":"allow","cost_usd":NaN}',
|
|
21
|
+
'{"outcome":"allow","layer":"judge","calls":[{"engine":"e","latency_ms":NaN}]}',
|
|
22
|
+
'{"outcome":"allow","layer":{"x":1},"engine":7,"source":[1],"p":"nope","calls":"x"}',
|
|
23
|
+
'{"outcome":"allow","latency_ms":Infinity,"layer":"judge","engine":"e"}',
|
|
24
|
+
],
|
|
25
|
+
)
|
|
26
|
+
def test_odd_log_values_never_break_load(tmp_path, line):
|
|
27
|
+
write(tmp_path, [json.dumps({"outcome": "hold", "layer": "rules"}), line])
|
|
28
|
+
data = load(tmp_path)
|
|
29
|
+
json.dumps(data, allow_nan=False) # strict JSON: no NaN/Infinity left
|
|
30
|
+
assert data["summary"]["outcomes"].get("hold") == 1
|
|
31
|
+
assert all(isinstance(k, str) or k is None for k in data["summary"]["outcomes"])
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def test_nan_latency_is_not_sampled(tmp_path):
|
|
35
|
+
write(
|
|
36
|
+
tmp_path,
|
|
37
|
+
[
|
|
38
|
+
'{"outcome":"allow","layer":"judge","calls":[{"engine":"e","latency_ms":NaN},{"engine":"e","latency_ms":5}]}'
|
|
39
|
+
],
|
|
40
|
+
)
|
|
41
|
+
e = load(tmp_path)["summary"]["engines"]["e"]
|
|
42
|
+
assert e["calls"] == 2 and e["p50_ms"] == 5 and not math.isnan(e["p50_ms"])
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def test_api_data_answers_with_a_bad_line(tmp_path):
|
|
46
|
+
from judgetap.dashboard.server import serve
|
|
47
|
+
|
|
48
|
+
write(tmp_path, ['{"outcome":"allow","cost_usd":"0.1"}', '{"outcome":["hold"]}'])
|
|
49
|
+
srv = serve(tmp_path, 0)
|
|
50
|
+
port = srv.server_address[1]
|
|
51
|
+
threading.Thread(target=srv.serve_forever, daemon=True).start()
|
|
52
|
+
try:
|
|
53
|
+
with urllib.request.urlopen(f"http://127.0.0.1:{port}/api/data") as res:
|
|
54
|
+
assert res.status == 200 and json.loads(res.read())["summary"]["total"] == 2
|
|
55
|
+
finally:
|
|
56
|
+
srv.shutdown()
|
|
57
|
+
srv.server_close()
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def test_port_zero_uses_the_bound_port(tmp_path):
|
|
61
|
+
from judgetap.dashboard.server import serve
|
|
62
|
+
|
|
63
|
+
srv = serve(tmp_path, 0)
|
|
64
|
+
port = srv.server_address[1]
|
|
65
|
+
assert port != 0
|
|
66
|
+
threading.Thread(target=srv.serve_forever, daemon=True).start()
|
|
67
|
+
try:
|
|
68
|
+
with urllib.request.urlopen(f"http://127.0.0.1:{port}/") as res:
|
|
69
|
+
assert res.status == 200
|
|
70
|
+
finally:
|
|
71
|
+
srv.shutdown()
|
|
72
|
+
srv.server_close()
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def test_keys_set_rejects_names_no_engine_reads(monkeypatch, capsys):
|
|
76
|
+
from judgetap.cli import main
|
|
77
|
+
|
|
78
|
+
asked = []
|
|
79
|
+
monkeypatch.setattr("getpass.getpass", lambda prompt: asked.append(prompt) or "v")
|
|
80
|
+
assert main(["keys", "set", "OPENAI_API_KEY"]) == 2
|
|
81
|
+
assert not asked and "isn't read by any engine" in capsys.readouterr().out
|
|
82
|
+
assert main(["keys", "status", "OPENAI_API_KEY"]) == 2
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def test_empty_new_env_wins_over_old(monkeypatch):
|
|
86
|
+
from judgetap._compat import env, env_source
|
|
87
|
+
|
|
88
|
+
monkeypatch.setenv("JUDGETAP_ENGINE", "")
|
|
89
|
+
monkeypatch.setenv("SNAPJUDGE_ENGINE", "llm")
|
|
90
|
+
assert env("ENGINE") is None and env_source("ENGINE") is None
|
|
91
|
+
monkeypatch.delenv("JUDGETAP_ENGINE")
|
|
92
|
+
assert env("ENGINE") == "llm" and env_source("ENGINE") == "$SNAPJUDGE_ENGINE"
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def test_nan_in_unknown_fields_is_scrubbed(tmp_path):
|
|
96
|
+
write(tmp_path, ['{"outcome":"allow","extra":{"x":[NaN, -Infinity]}}'])
|
|
97
|
+
json.dumps(load(tmp_path), allow_nan=False)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def test_huge_int_cost_keeps_the_record(tmp_path):
|
|
101
|
+
import json
|
|
102
|
+
|
|
103
|
+
from judgetap.dashboard.data import load
|
|
104
|
+
|
|
105
|
+
(tmp_path / "guard.jsonl").write_text(
|
|
106
|
+
json.dumps({"outcome": "hold", "cost_usd": 10**400}) + "\n"
|
|
107
|
+
)
|
|
108
|
+
s = load(tmp_path)["summary"]
|
|
109
|
+
assert s["total"] == 1 and s["outcomes"] == {"hold": 1}
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def test_invalid_legacy_latency_adds_no_sample(tmp_path):
|
|
113
|
+
from judgetap.dashboard.data import load
|
|
114
|
+
|
|
115
|
+
(tmp_path / "guard.jsonl").write_text(
|
|
116
|
+
'{"outcome":"allow","layer":"judge","engine":"e","latency_ms":NaN}\n'
|
|
117
|
+
)
|
|
118
|
+
e = load(tmp_path)["summary"]["engines"]["e"]
|
|
119
|
+
assert e["calls"] == 1 and e["p50_ms"] is None
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def test_overflowing_cost_total_still_serves(tmp_path):
|
|
123
|
+
import json
|
|
124
|
+
import socket
|
|
125
|
+
import threading
|
|
126
|
+
import urllib.request
|
|
127
|
+
from http.server import ThreadingHTTPServer
|
|
128
|
+
|
|
129
|
+
from judgetap.dashboard.server import make_handler
|
|
130
|
+
|
|
131
|
+
(tmp_path / "guard.jsonl").write_text(
|
|
132
|
+
"\n".join(json.dumps({"outcome": "allow", "cost_usd": 1e308}) for _ in range(2))
|
|
133
|
+
+ "\n"
|
|
134
|
+
)
|
|
135
|
+
with socket.socket() as sock:
|
|
136
|
+
sock.bind(("127.0.0.1", 0))
|
|
137
|
+
port = sock.getsockname()[1]
|
|
138
|
+
srv = ThreadingHTTPServer(("127.0.0.1", port), make_handler(tmp_path, "tok", port))
|
|
139
|
+
threading.Thread(target=srv.serve_forever, daemon=True).start()
|
|
140
|
+
try:
|
|
141
|
+
with urllib.request.urlopen(f"http://127.0.0.1:{port}/api/data") as res:
|
|
142
|
+
assert res.status == 200
|
|
143
|
+
assert json.loads(res.read())["summary"]["cost_usd"] is None
|
|
144
|
+
finally:
|
|
145
|
+
srv.shutdown()
|
|
146
|
+
srv.server_close()
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def test_malformed_calls_field_keeps_legacy_fallback(tmp_path):
|
|
150
|
+
import json
|
|
151
|
+
|
|
152
|
+
from judgetap.dashboard.data import load
|
|
153
|
+
|
|
154
|
+
(tmp_path / "guard.jsonl").write_text(
|
|
155
|
+
json.dumps(
|
|
156
|
+
{
|
|
157
|
+
"id": "a",
|
|
158
|
+
"outcome": "allow",
|
|
159
|
+
"layer": "judge",
|
|
160
|
+
"engine": "e",
|
|
161
|
+
"latency_ms": 12,
|
|
162
|
+
"calls": "x",
|
|
163
|
+
}
|
|
164
|
+
)
|
|
165
|
+
+ "\n"
|
|
166
|
+
)
|
|
167
|
+
e = load(tmp_path)["summary"]["engines"]["e"]
|
|
168
|
+
assert e["calls"] == 1 and e["p50_ms"] == 12
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def test_deeply_nested_extra_field_does_not_drop_the_record(tmp_path):
|
|
172
|
+
from judgetap.dashboard.data import load
|
|
173
|
+
|
|
174
|
+
deep = "[" * 5000 + "]" * 5000
|
|
175
|
+
line = '{"id": "a", "outcome": "hold", "layer": "rules", "extra": ' + deep + "}"
|
|
176
|
+
(tmp_path / "guard.jsonl").write_text(line + "\n")
|
|
177
|
+
assert load(tmp_path)["summary"]["total"] == 1
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|