judgetap 0.0.2.dev26__tar.gz → 0.0.2.dev28__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/PKG-INFO +2 -2
  2. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/README.md +1 -1
  3. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/docs/SPEC.md +1 -1
  4. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/pyproject.toml +1 -1
  5. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/__init__.py +1 -1
  6. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/_compat.py +8 -6
  7. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/cli.py +16 -4
  8. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/dashboard/data.py +97 -6
  9. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/dashboard/server.py +9 -1
  10. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/guard/install.py +13 -2
  11. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/guard/loop.py +87 -3
  12. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/guard/stop.py +8 -0
  13. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/secrets.py +3 -0
  14. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_guard_loop.py +163 -8
  15. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_guard_stop.py +37 -0
  16. judgetap-0.0.2.dev28/tests/test_robustness_68.py +177 -0
  17. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/.github/workflows/ci.yml +0 -0
  18. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/.github/workflows/demo.yml +0 -0
  19. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/.github/workflows/release.yml +0 -0
  20. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/.gitignore +0 -0
  21. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/.python-version +0 -0
  22. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/.release-please-manifest.json +0 -0
  23. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/CONTRIBUTING.md +0 -0
  24. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/LICENSE +0 -0
  25. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/docs/demo.tape +0 -0
  26. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/release-please-config.json +0 -0
  27. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/api.py +0 -0
  28. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/cascade.py +0 -0
  29. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/dashboard/__init__.py +0 -0
  30. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/dashboard/page.html +0 -0
  31. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/decision_log.py +0 -0
  32. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/engine.py +0 -0
  33. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/engines/__init__.py +0 -0
  34. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/engines/agentjev.py +0 -0
  35. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/engines/jev.py +0 -0
  36. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/engines/laya.py +0 -0
  37. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/engines/llm.py +0 -0
  38. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/errors.py +0 -0
  39. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/evaluate.py +0 -0
  40. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/guard/__init__.py +0 -0
  41. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/guard/core.py +0 -0
  42. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/guard/hook.py +0 -0
  43. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/guard/rules.py +0 -0
  44. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/py.typed +0 -0
  45. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/testing.py +0 -0
  46. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/src/judgetap/types.py +0 -0
  47. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_api.py +0 -0
  48. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_call_accounting.py +0 -0
  49. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_calls.py +0 -0
  50. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_cascade.py +0 -0
  51. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_dashboard.py +0 -0
  52. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_decision_log.py +0 -0
  53. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_engine_jev.py +0 -0
  54. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_engine_llm.py +0 -0
  55. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_engine_local.py +0 -0
  56. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_evaluate.py +0 -0
  57. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_guard_agents.py +0 -0
  58. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_guard_core.py +0 -0
  59. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_guard_hook.py +0 -0
  60. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_guard_rules.py +0 -0
  61. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_questions.py +0 -0
  62. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/tests/test_secrets.py +0 -0
  63. {judgetap-0.0.2.dev26 → judgetap-0.0.2.dev28}/uv.lock +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: judgetap
3
- Version: 0.0.2.dev26
3
+ Version: 0.0.2.dev28
4
4
  Summary: Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development.
5
5
  Project-URL: Homepage, https://github.com/mergesafe-ai/judgetap
6
6
  Author-email: Omer Bar-Ness <omer@zsquared.io>
@@ -80,7 +80,7 @@ On a CPU-only Linux box, install the CPU PyTorch wheel before `judgetap[laya]` (
80
80
 
81
81
  ## Guard details
82
82
 
83
- **Loop detection (Claude Code).** A `PostToolUse` hook (`judgetap guard post`, added by `guard install --for claude-code`) notices when the same command or edit fails the same way three times within eight actions and adds a note asking the agent to re-plan. It never blocks, uses no model, and stores only a redacted action and an error hash per session. Cursor and Codex: not yet.
83
+ **Loop detection (Claude Code).** `PostToolUse` and `PostToolUseFailure` hooks (`judgetap guard post`, added by `guard install --for claude-code`) notices when the same command or edit fails the same way three times within eight actions and adds a note asking the agent to re-plan. It never blocks, uses no model, and stores only a redacted action and an error hash per session. Cursor and Codex: not yet.
84
84
 
85
85
  - **Agents:** Claude Code (shell, writes and edits), Cursor and Codex (shell only; their hooks don't expose writes and edits).
86
86
  - **Engine:** `judgetap guard install` uses one you already have (`$JUDGETAP_ENGINE`, a `TYPESAFE_API_KEY`, or a local AgentJev) and saves it in `~/.judgetap/guard.toml`, because agents often run hooks without your shell's environment.
@@ -63,7 +63,7 @@ On a CPU-only Linux box, install the CPU PyTorch wheel before `judgetap[laya]` (
63
63
 
64
64
  ## Guard details
65
65
 
66
- **Loop detection (Claude Code).** A `PostToolUse` hook (`judgetap guard post`, added by `guard install --for claude-code`) notices when the same command or edit fails the same way three times within eight actions and adds a note asking the agent to re-plan. It never blocks, uses no model, and stores only a redacted action and an error hash per session. Cursor and Codex: not yet.
66
+ **Loop detection (Claude Code).** `PostToolUse` and `PostToolUseFailure` hooks (`judgetap guard post`, added by `guard install --for claude-code`) notices when the same command or edit fails the same way three times within eight actions and adds a note asking the agent to re-plan. It never blocks, uses no model, and stores only a redacted action and an error hash per session. Cursor and Codex: not yet.
67
67
 
68
68
  - **Agents:** Claude Code (shell, writes and edits), Cursor and Codex (shell only; their hooks don't expose writes and edits).
69
69
  - **Engine:** `judgetap guard install` uses one you already have (`$JUDGETAP_ENGINE`, a `TYPESAFE_API_KEY`, or a local AgentJev) and saves it in `~/.judgetap/guard.toml`, because agents often run hooks without your shell's environment.
@@ -67,7 +67,7 @@ A pre-action hook for coding agents, built on the core.
67
67
  - **Outcomes**: allow (silent), hold (block with a reason the agent reads and re-plans from), ask (escalate to the user). Holds should be rare; the target is under 5 per 1,000 calls.
68
68
  - **Fails safe and visibly**: Claude Code treats a crashing hook as non-blocking, so the guard catches its own errors, applies the rules layer alone, and says so.
69
69
  - **Log**: every decision to a local JSONL, so `judgetap guard stats` can report holds and cost. Marking a hold as a false alarm comes with the dashboard (#10).
70
- - **Loop detection** (Claude Code `PostToolUse`, no model): the same action failing with the same error (numbers ignored) 3 times in the last 8 actions adds `additionalContext` telling the agent to re-plan; a success of that action resets the count. Never blocks. Per-session ring buffer of 20 redacted actions and error hashes in `~/.judgetap/sessions/`, 0600. Logged as layer `loop`, outcome `note`.
70
+ - **Loop detection** (Claude Code `PostToolUseFailure` for failures, `PostToolUse` for successes that reset a streak; no model): the same action failing with the same error (numbers ignored) 3 times in the last 8 actions adds `additionalContext` telling the agent to re-plan; a success of that action resets the count. Never blocks. Per-session ring buffer of 20 redacted actions and error hashes in `~/.judgetap/sessions/`, 0600. Logged as layer `loop`, outcome `note`.
71
71
 
72
72
  ## Decided: the guard's default engine (#8)
73
73
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "judgetap"
3
- version = "0.0.2.dev26"
3
+ version = "0.0.2.dev28"
4
4
  description = "Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development."
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -23,7 +23,7 @@ from judgetap.errors import (
23
23
  )
24
24
  from judgetap.types import Decision, Question
25
25
 
26
- __version__ = "0.0.2.dev26" # x-release-please-version
26
+ __version__ = "0.0.2.dev28" # x-release-please-version
27
27
 
28
28
  __all__ = [
29
29
  "Cascade",
@@ -14,8 +14,11 @@ OLD_PREFIX, NEW_PREFIX = "SNAPJUDGE_", "JUDGETAP_"
14
14
 
15
15
 
16
16
  def env(name: str) -> str | None:
17
- """$JUDGETAP_<name>, else the old $SNAPJUDGE_<name>."""
18
- return os.environ.get(NEW_PREFIX + name) or os.environ.get(OLD_PREFIX + name)
17
+ """$JUDGETAP_<name>, else the old $SNAPJUDGE_<name>. The new name wins
18
+ whenever it is set, even to an empty string (which reads as unset)."""
19
+ if NEW_PREFIX + name in os.environ:
20
+ return os.environ[NEW_PREFIX + name] or None
21
+ return os.environ.get(OLD_PREFIX + name) or None
19
22
 
20
23
 
21
24
  def default_home() -> Path:
@@ -26,7 +29,6 @@ def default_home() -> Path:
26
29
 
27
30
  def env_source(name: str) -> str | None:
28
31
  """Which variable `env(name)` read: '$JUDGETAP_<name>' or '$SNAPJUDGE_<name>'."""
29
- for prefix in (NEW_PREFIX, OLD_PREFIX):
30
- if os.environ.get(prefix + name):
31
- return f"${prefix}{name}"
32
- return None
32
+ if NEW_PREFIX + name in os.environ: # set, even empty: the old name is ignored
33
+ return f"${NEW_PREFIX}{name}" if os.environ[NEW_PREFIX + name] else None
34
+ return f"${OLD_PREFIX}{name}" if os.environ.get(OLD_PREFIX + name) else None
@@ -190,7 +190,7 @@ def _dashboard(args) -> int:
190
190
  from judgetap.guard.hook import home
191
191
 
192
192
  server = serve(home(), args.port)
193
- url = f"http://127.0.0.1:{args.port}/"
193
+ url = f"http://127.0.0.1:{server.server_address[1]}/" # the bound port (--port 0)
194
194
  print(f"judgetap dashboard on {url} (Ctrl+C to stop); reading {home()}")
195
195
  if not args.no_browser:
196
196
  webbrowser.open(url)
@@ -238,8 +238,14 @@ def _offer_keychain(stdin=None) -> None:
238
238
  def _keys_set(args) -> int:
239
239
  import getpass
240
240
 
241
- from judgetap.secrets import set_key
241
+ from judgetap.secrets import KNOWN_KEYS, set_key
242
242
 
243
+ if args.name not in KNOWN_KEYS:
244
+ print(
245
+ f"{args.name} isn't read by any engine; known keys: {', '.join(KNOWN_KEYS)}."
246
+ )
247
+ print("The llm engine reads its provider's key from the environment only.")
248
+ return 2
243
249
  value = getpass.getpass(f"{args.name}: ") # never echoed, never in argv
244
250
  if not value:
245
251
  print("Nothing entered; not saved.")
@@ -254,9 +260,15 @@ def _keys_set(args) -> int:
254
260
 
255
261
 
256
262
  def _keys_status(args) -> int:
257
- from judgetap.secrets import key_source
263
+ from judgetap.secrets import KNOWN_KEYS, key_source
258
264
 
259
- for name in args.names or ["TYPESAFE_API_KEY"]:
265
+ unknown = [n for n in args.names if n not in KNOWN_KEYS]
266
+ if unknown:
267
+ print(
268
+ f"Not read by any engine: {', '.join(unknown)}; known keys: {', '.join(KNOWN_KEYS)}."
269
+ )
270
+ return 2
271
+ for name in args.names or KNOWN_KEYS:
260
272
  print(f"{name}: {key_source(name) or 'not set'}")
261
273
  return 0
262
274
 
@@ -85,8 +85,12 @@ def _load(home: Path, today: date) -> dict[str, Any]:
85
85
  calls: Counter = Counter() # true totals; the latency deques are capped
86
86
  recent: deque = deque(maxlen=MAX_RECENT)
87
87
  cost, total, false_holds, library, loops, stops = 0.0, 0, 0, 0, 0, 0
88
- for n, r in _iter_jsonl(home / "guard.jsonl"):
89
- r["id"] = record_id(r, n)
88
+ for n, raw in _iter_jsonl(home / "guard.jsonl"):
89
+ try:
90
+ r = _clean(raw)
91
+ r["id"] = record_id(r, n)
92
+ except Exception: # noqa: BLE001, S112 -- one unreadable record never breaks the page
93
+ continue
90
94
  r["source"] = r.get("source") or "guard"
91
95
  r["false_alarm"] = r["id"] in false_alarms
92
96
  cost += r.get("cost_usd") or 0
@@ -118,7 +122,8 @@ def _load(home: Path, today: date) -> dict[str, Any]:
118
122
  "outcomes": dict(outcomes),
119
123
  "holds_per_1000": round(1000 * outcomes["hold"] / total, 1) if total else None,
120
124
  "false_alarms": false_holds,
121
- "cost_usd": round(cost, 6),
125
+ # Individually finite costs can still sum past float range.
126
+ "cost_usd": round(cost, 6) if math.isfinite(cost) else None,
122
127
  "engines": {
123
128
  name: {
124
129
  "calls": calls[name],
@@ -132,6 +137,89 @@ def _load(home: Path, today: date) -> dict[str, Any]:
132
137
  return {"summary": summary, "recent": list(recent)[::-1]}
133
138
 
134
139
 
140
+ STR_FIELDS = (
141
+ "id",
142
+ "ts",
143
+ "session",
144
+ "source",
145
+ "tool",
146
+ "subject",
147
+ "outcome",
148
+ "layer",
149
+ "rule",
150
+ "reason",
151
+ "engine",
152
+ "error",
153
+ "call",
154
+ "batch",
155
+ )
156
+ NUM_FIELDS = ("cost_usd", "latency_ms")
157
+
158
+
159
+ def _finite(value: Any) -> float | None:
160
+ """A finite int/float as float; anything else (str, bool, NaN, inf) is None."""
161
+ if isinstance(value, bool) or not isinstance(value, int | float):
162
+ return None
163
+ try:
164
+ number = float(value) # huge JSON ints overflow here: an invalid field
165
+ except OverflowError:
166
+ return None
167
+ return number if math.isfinite(number) else None
168
+
169
+
170
+ MAX_DEPTH = 8 # deeper than any field the page reads
171
+
172
+
173
+ def _scrub(value: Any, depth: int = 0) -> Any:
174
+ """NaN/Infinity anywhere (even in fields the page doesn't use) become None.
175
+ Nesting past MAX_DEPTH is cut to None, so a pathological unrelated field
176
+ can't exhaust the stack and cost the record."""
177
+ if isinstance(value, float) and not math.isfinite(value):
178
+ return None
179
+ if isinstance(value, dict | list) and depth >= MAX_DEPTH:
180
+ return None
181
+ if isinstance(value, dict):
182
+ return {k: _scrub(v, depth + 1) for k, v in value.items()}
183
+ if isinstance(value, list):
184
+ return [_scrub(v, depth + 1) for v in value]
185
+ return value
186
+
187
+
188
+ def _clean(r: dict[str, Any]) -> dict[str, Any]:
189
+ """The log is ours but may be edited, truncated or written by an older
190
+ version: coerce every field the page uses to its expected type, so no
191
+ value can crash the summary or make the JSON invalid (NaN)."""
192
+ out = _scrub(dict(r))
193
+ for key in STR_FIELDS:
194
+ if key in out and not isinstance(out[key], str):
195
+ out[key] = None
196
+ for key in NUM_FIELDS:
197
+ if key in out:
198
+ out[key] = _finite(out[key])
199
+ if not isinstance(out.get("p"), dict):
200
+ out["p"] = {}
201
+ else:
202
+ out["p"] = {
203
+ k: v
204
+ for k, v in out["p"].items()
205
+ if isinstance(k, str) and _finite(v) is not None
206
+ }
207
+ if "calls" in out and not isinstance(out["calls"], list):
208
+ # Not a calls list at all: treat the record as having none, so the
209
+ # legacy fallback still counts its engine.
210
+ del out["calls"]
211
+ if "calls" in out:
212
+ # Calls are echoed back in "recent" too: keep only well-formed ones
213
+ # with a finite (or absent) latency.
214
+ calls = out["calls"]
215
+ out["calls"] = [
216
+ {**c, "latency_ms": _finite(c.get("latency_ms"))}
217
+ for c in calls
218
+ if isinstance(c, dict) and isinstance(c.get("engine"), str)
219
+ ]
220
+ return out
221
+
222
+
135
223
  def _count_calls(r: dict[str, Any], calls: Counter, by_engine) -> None:
136
224
  """Engine metrics come from the calls each record reports, each with its
137
225
  own latency: nothing is inferred. A batch writes its calls on one record
@@ -143,14 +231,17 @@ def _count_calls(r: dict[str, Any], calls: Counter, by_engine) -> None:
143
231
  for c in reported:
144
232
  if isinstance(c, dict) and isinstance(c.get("engine"), str):
145
233
  calls[c["engine"]] += 1
146
- if isinstance(c.get("latency_ms"), int | float):
147
- by_engine[c["engine"]].append(float(c["latency_ms"]))
234
+ latency = _finite(c.get("latency_ms"))
235
+ if latency is not None:
236
+ by_engine[c["engine"]].append(latency)
148
237
  return
149
238
  if not r.get("engine") or r.get("error"):
150
239
  return
151
240
  if r.get("layer") == "judge":
152
241
  calls[r["engine"]] += 1
153
- by_engine[r["engine"]].append(float(r.get("latency_ms") or 0))
242
+ latency = _finite(r.get("latency_ms"))
243
+ if latency is not None: # no sample for a missing or invalid latency
244
+ by_engine[r["engine"]].append(latency)
154
245
  elif r.get("layer") == "stop" and r.get("outcome") in ("allow", "block"):
155
246
  # A Stop check from before calls were logged: it asked its engine
156
247
  # once, but recorded no latency, so only the call is counted.
@@ -44,7 +44,10 @@ def make_handler(home: Path, token: str, port: int) -> type[BaseHTTPRequestHandl
44
44
  self.wfile.write(body)
45
45
 
46
46
  def _json(self, status: int, obj: object) -> None:
47
- self._send(status, json.dumps(obj).encode(), "application/json")
47
+ # allow_nan=False: bare NaN isn't JSON and would break the page.
48
+ self._send(
49
+ status, json.dumps(obj, allow_nan=False).encode(), "application/json"
50
+ )
48
51
 
49
52
  def do_GET(self) -> None:
50
53
  if not self._host_ok():
@@ -94,4 +97,9 @@ def make_handler(home: Path, token: str, port: int) -> type[BaseHTTPRequestHandl
94
97
  def serve(home: Path, port: int = 8765) -> ThreadingHTTPServer:
95
98
  token = secrets.token_urlsafe(24)
96
99
  server = ThreadingHTTPServer(("127.0.0.1", port), make_handler(home, token, port))
100
+ # Port 0 means "any free port": the Host allow-list must use the port
101
+ # the OS actually bound, or every request is refused.
102
+ bound = server.server_address[1]
103
+ if bound != port:
104
+ server.RequestHandlerClass = make_handler(home, token, bound)
97
105
  return server
@@ -89,7 +89,14 @@ def _event(agent: str) -> str:
89
89
 
90
90
  def _events(agent: str, with_stop: bool = False) -> list[str]:
91
91
  if agent == "claude-code":
92
- return ["PreToolUse", "PostToolUse", *(["Stop"] if with_stop else [])]
92
+ # Loop detection needs both: failures arrive only on
93
+ # PostToolUseFailure, successes (which end a streak) on PostToolUse.
94
+ return [
95
+ "PreToolUse",
96
+ "PostToolUse",
97
+ "PostToolUseFailure",
98
+ *(["Stop"] if with_stop else []),
99
+ ]
93
100
  return [_event(agent)]
94
101
 
95
102
 
@@ -98,7 +105,11 @@ def _entry(agent: str, event: str) -> dict:
98
105
  return {"command": hook_command(agent)}
99
106
  if event == "Stop": # Stop takes no matcher
100
107
  return {"hooks": [{"type": "command", "command": STOP_COMMAND}]}
101
- command = POST_COMMAND if event == "PostToolUse" else hook_command(agent)
108
+ command = (
109
+ POST_COMMAND
110
+ if event in ("PostToolUse", "PostToolUseFailure")
111
+ else hook_command(agent)
112
+ )
102
113
  # Codex's PreToolUse fires for shell only today; the matcher says so.
103
114
  matcher = "^(exec_command|shell|Bash)$" if agent == "codex" else MATCHER
104
115
  return {"matcher": matcher, "hooks": [{"type": "command", "command": command}]}
@@ -13,6 +13,7 @@ import json
13
13
  import os
14
14
  import re
15
15
  import sys
16
+ import time
16
17
  import uuid
17
18
  from contextlib import contextmanager
18
19
  from datetime import UTC, datetime
@@ -50,10 +51,20 @@ EXIT_PREFIX = re.compile(r"^\s*exit code[: ]\s*(-?\d+)", re.IGNORECASE)
50
51
 
51
52
 
52
53
  def failure(payload: dict[str, Any]) -> str | None:
53
- """A short hash of the error, or None when the call succeeded."""
54
+ """A short hash of the error, or None when the call succeeded.
55
+
56
+ Claude Code sends failures on PostToolUseFailure with the error as a
57
+ top-level `error` string (for Bash, first line "Exit code N"); successes
58
+ arrive on PostToolUse. The tool_response checks cover other agents and
59
+ older shapes."""
54
60
  resp = payload.get("tool_response")
55
61
  error = payload.get("error")
56
- text, failed, code = "", bool(error), None
62
+ failure_event = payload.get("hook_event_name") == "PostToolUseFailure"
63
+ text, failed, code = "", bool(error) or failure_event, None
64
+ if isinstance(error, str):
65
+ m = EXIT_PREFIX.match(error)
66
+ if m:
67
+ code = int(m.group(1))
57
68
  if isinstance(resp, dict):
58
69
  code = resp.get("exit_code", resp.get("exitCode", resp.get("returncode")))
59
70
  if isinstance(code, int) and code != 0:
@@ -80,6 +91,72 @@ def failure(payload: dict[str, Any]) -> str | None:
80
91
  return hashlib.sha256(stable.encode()).hexdigest()[:12]
81
92
 
82
93
 
94
+ SESSION_TTL_SECONDS = 7 * 24 * 3600
95
+ PRUNE_EVERY_SECONDS = 3600
96
+
97
+
98
+ def prune_sessions(directory: Path, now: float | None = None) -> None:
99
+ """Delete session state (json, stop, tmp) untouched for SESSION_TTL_SECONDS.
100
+
101
+ Lock files are never deleted: unlinking a lock someone holds would let a
102
+ second hook lock a fresh inode and break mutual exclusion. They are empty,
103
+ so keeping them costs an inode, not space. A session's data is only
104
+ removed while holding its lock without waiting; a busy session is skipped.
105
+ Runs at most once per PRUNE_EVERY_SECONDS and never raises."""
106
+ try:
107
+ now = time.time() if now is None else now
108
+ marker = directory / ".pruned"
109
+ if marker.exists() and now - marker.stat().st_mtime < PRUNE_EVERY_SECONDS:
110
+ return
111
+ directory.mkdir(mode=0o700, parents=True, exist_ok=True)
112
+ marker.touch()
113
+ os.utime(marker, (now, now))
114
+ for f in directory.iterdir():
115
+ if f.name == ".pruned" or f.suffix not in (".json", ".stop", ".tmp"):
116
+ continue
117
+ try:
118
+ if now - f.stat().st_mtime <= SESSION_TTL_SECONDS:
119
+ continue
120
+ _unlink_if_unlocked(f)
121
+ except OSError:
122
+ continue
123
+ except OSError:
124
+ return
125
+
126
+
127
+ TMP_NAME = re.compile(r"^(?P<stem>.+)\.[0-9a-f]{32}\.tmp$")
128
+
129
+
130
+ def _lock_for(f: Path) -> Path:
131
+ """The lock _session_lock takes for this file. State files are
132
+ `<id>.json` / `<id>.stop` and lock `<id>.lock` (with_suffix, so a
133
+ session id may itself contain dots); their temp files are
134
+ `<id>.<uuid>.tmp` (with_suffix replaces .json/.stop), so the stem before
135
+ the uuid is the session id."""
136
+ m = TMP_NAME.match(f.name)
137
+ if m:
138
+ return f.with_name(m.group("stem") + ".lock")
139
+ return f.with_suffix(".lock")
140
+
141
+
142
+ def _unlink_if_unlocked(f: Path) -> None:
143
+ """Remove f only while holding its session lock (non-blocking)."""
144
+ lock = _lock_for(f)
145
+ fd = os.open(lock, os.O_WRONLY | os.O_CREAT, 0o600)
146
+ try:
147
+ try:
148
+ import fcntl
149
+
150
+ fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
151
+ except ImportError:
152
+ pass # no flock (Windows): best effort, as for the hooks themselves
153
+ except OSError:
154
+ return # a hook holds it: this session is in use, keep its data
155
+ f.unlink(missing_ok=True)
156
+ finally:
157
+ os.close(fd)
158
+
159
+
83
160
  def _state_path(session: str | None) -> Path | None:
84
161
  if not session or not SESSION_ID.fullmatch(session):
85
162
  return None
@@ -167,6 +244,9 @@ def handle(payload: dict[str, Any]) -> dict[str, Any] | None:
167
244
  path = _state_path(payload.get("session_id"))
168
245
  if act is None or path is None:
169
246
  return None
247
+ if payload.get("is_interrupt"):
248
+ return None # an abort, not an error the tool reported: not a loop signal
249
+ prune_sessions(path.parent)
170
250
  # Only a hash and the (redacted) action are stored: never file contents.
171
251
  # Hooks for one session can finish together: serialise the
172
252
  # read-append-write so no action is lost.
@@ -187,9 +267,13 @@ def handle(payload: dict[str, Any]) -> dict[str, Any] | None:
187
267
  return None
188
268
  _log(payload.get("session_id"), actions[-1]["act"], count)
189
269
  shown = actions[-1]["act"].split(":", 1)[1][:200]
270
+ event = payload.get("hook_event_name")
271
+ if event not in ("PostToolUse", "PostToolUseFailure"):
272
+ event = "PostToolUseFailure"
190
273
  return {
191
274
  "hookSpecificOutput": {
192
- "hookEventName": "PostToolUse",
275
+ # Answer on the event that fired (PostToolUseFailure for failures).
276
+ "hookEventName": event,
193
277
  "additionalContext": (
194
278
  f"judgetap: `{shown}` has now failed the same way {count} times; "
195
279
  "stop and re-plan (read the error, try a different approach)."
@@ -113,6 +113,9 @@ def _count_path(session: str) -> Path:
113
113
  def _take_block(session: str) -> bool:
114
114
  """Count one block for this session; False once the cap is reached."""
115
115
  path = _count_path(session)
116
+ from judgetap.guard.loop import prune_sessions
117
+
118
+ prune_sessions(path.parent)
116
119
  with _session_lock(path):
117
120
  try:
118
121
  count = int(path.read_text().strip() or 0)
@@ -177,6 +180,11 @@ def handle(payload: dict[str, Any], engine=None) -> dict[str, Any] | None:
177
180
  return None # needs a judge: no rules-only behaviour here
178
181
  transcript = payload.get("transcript_path")
179
182
  task, said = task_and_reply(transcript)
183
+ # Claude Code passes the final reply directly: the transcript is written
184
+ # asynchronously and may not contain it yet at Stop time.
185
+ last = payload.get("last_assistant_message")
186
+ if isinstance(last, str) and last.strip():
187
+ said = last[-ASSISTANT_TAIL_CHARS:]
180
188
  if not task or not said:
181
189
  _log(
182
190
  session,
@@ -10,6 +10,9 @@ from __future__ import annotations
10
10
 
11
11
  import os
12
12
 
13
+ # Keys an engine reads through get_key. Saving any other name to the keychain
14
+ # would store something nothing reads.
15
+ KNOWN_KEYS = ("TYPESAFE_API_KEY",)
13
16
  SERVICE = "judgetap"
14
17
  OLD_SERVICE = "snapjudge" # keys saved before the rename (#38)
15
18
 
@@ -16,16 +16,33 @@ def _home(tmp_path, monkeypatch):
16
16
  def call(
17
17
  command, *, fail=True, err="npm ERR! missing script: build", session="s1", code=1
18
18
  ):
19
- payload = {
19
+ """A documented Claude Code payload: failures on PostToolUseFailure with a
20
+ top-level `error` ("Exit code N" first line for Bash), successes on
21
+ PostToolUse with a tool_response."""
22
+ base = {
20
23
  "session_id": session,
24
+ "transcript_path": "/tmp/t.jsonl",
25
+ "cwd": "/tmp",
26
+ "permission_mode": "default",
21
27
  "tool_name": "Bash",
22
- "tool_input": {"command": command},
23
- "tool_response": {
24
- "stdout": "",
25
- "stderr": err if fail else "",
26
- "exit_code": code if fail else 0,
27
- },
28
+ "tool_input": {"command": command, "description": "run"},
29
+ "tool_use_id": "toolu_01ABC",
28
30
  }
31
+ if fail:
32
+ payload = {
33
+ **base,
34
+ "hook_event_name": "PostToolUseFailure",
35
+ "error": f"Exit code {code}\n{err}",
36
+ "is_interrupt": False,
37
+ "duration_ms": 10,
38
+ }
39
+ else:
40
+ payload = {
41
+ **base,
42
+ "hook_event_name": "PostToolUse",
43
+ "tool_response": {"stdout": "ok", "stderr": "", "interrupted": False},
44
+ "duration_ms": 10,
45
+ }
29
46
  out = io.StringIO()
30
47
  assert loop.run(io.StringIO(json.dumps(payload)), out) == 0
31
48
  return json.loads(out.getvalue()) if out.getvalue() else None
@@ -36,7 +53,7 @@ def test_third_identical_failure_adds_a_note():
36
53
  assert call("npm run build") is None
37
54
  out = call("npm run build")
38
55
  ctx = out["hookSpecificOutput"]
39
- assert ctx["hookEventName"] == "PostToolUse"
56
+ assert ctx["hookEventName"] == "PostToolUseFailure"
40
57
  assert (
41
58
  "`npm run build` has now failed the same way 3 times"
42
59
  in ctx["additionalContext"]
@@ -225,3 +242,141 @@ def test_concurrent_hooks_do_not_lose_actions(tmp_path, monkeypatch):
225
242
  for t in threads:
226
243
  t.join()
227
244
  assert len(real_load(tmp_path / "sessions" / "race.json")) == 6
245
+
246
+
247
+ DOC_FAILURE = {
248
+ # Verbatim from code.claude.com/docs/en/hooks#posttoolusefailure-input
249
+ "session_id": "abc123",
250
+ "transcript_path": "/Users/.../.claude/projects/.../00893aaf-19fa-41d2-8238-13269b9b3ca0.jsonl",
251
+ "cwd": "/Users/...",
252
+ "permission_mode": "default",
253
+ "hook_event_name": "PostToolUseFailure",
254
+ "tool_name": "Bash",
255
+ "tool_input": {"command": "npm test", "description": "Run test suite"},
256
+ "tool_use_id": "toolu_01ABC123...",
257
+ "error": "Exit code 1\nError: Cannot find module 'express'",
258
+ "is_interrupt": False,
259
+ "duration_ms": 4187,
260
+ }
261
+
262
+
263
+ def test_documented_failure_payload_triggers_on_the_third_repeat():
264
+ outs = [loop.handle(dict(DOC_FAILURE)) for _ in range(3)]
265
+ assert outs[:2] == [None, None]
266
+ assert outs[2]["hookSpecificOutput"]["hookEventName"] == "PostToolUseFailure"
267
+
268
+
269
+ def test_interrupts_are_not_loop_signals():
270
+ for _ in range(4):
271
+ assert loop.handle({**DOC_FAILURE, "is_interrupt": True}) is None
272
+
273
+
274
+ def test_install_adds_post_tool_use_failure_and_upgrades_old_installs(tmp_path):
275
+ path = tmp_path / "settings.json"
276
+ old = {
277
+ "hooks": {
278
+ "PostToolUse": [
279
+ {
280
+ "matcher": "Bash",
281
+ "hooks": [{"type": "command", "command": POST_COMMAND}],
282
+ }
283
+ ]
284
+ }
285
+ }
286
+ path.write_text(json.dumps(old))
287
+ assert install(path) is True
288
+ hooks = json.loads(path.read_text())["hooks"]
289
+ assert POST_COMMAND in json.dumps(hooks["PostToolUseFailure"])
290
+ assert install(path) is False
291
+ assert uninstall(path) is True
292
+ assert "PostToolUseFailure" not in json.loads(path.read_text()).get("hooks", {})
293
+
294
+
295
+ def test_prune_sessions_removes_old_state_once_an_hour(tmp_path):
296
+ import os
297
+
298
+ d = tmp_path / "sessions"
299
+ d.mkdir()
300
+ old, fresh = d / "a.json", d / "b.json"
301
+ old.write_text("[]")
302
+ fresh.write_text("[]")
303
+ now = 10_000_000.0
304
+ os.utime(old, (now - 8 * 86400, now - 8 * 86400))
305
+ os.utime(fresh, (now - 60, now - 60))
306
+ loop.prune_sessions(d, now=now)
307
+ assert not old.exists() and fresh.exists()
308
+ stale = d / "c.json"
309
+ stale.write_text("[]")
310
+ os.utime(stale, (now - 9 * 86400, now - 9 * 86400))
311
+ loop.prune_sessions(d, now=now + 60) # within the hour: no sweep
312
+ assert stale.exists()
313
+ loop.prune_sessions(d, now=now + 3700)
314
+ assert not stale.exists()
315
+
316
+
317
+ def test_prune_keeps_locks_and_skips_busy_sessions(tmp_path):
318
+ import fcntl
319
+ import os
320
+ import time
321
+
322
+ from judgetap.guard.loop import SESSION_TTL_SECONDS, prune_sessions
323
+
324
+ old = time.time() - SESSION_TTL_SECONDS - 100
325
+ for name in (
326
+ "idle.json",
327
+ "idle.lock",
328
+ "busy.json",
329
+ "busy.lock",
330
+ "idle.stop",
331
+ "x." + "ab" * 16 + ".tmp",
332
+ ):
333
+ p = tmp_path / name
334
+ p.write_text("{}")
335
+ os.utime(p, (old, old))
336
+ fd = os.open(tmp_path / "busy.lock", os.O_WRONLY)
337
+ fcntl.flock(fd, fcntl.LOCK_EX) # another hook is working on "busy"
338
+ try:
339
+ prune_sessions(tmp_path)
340
+ finally:
341
+ os.close(fd)
342
+ left = sorted(p.name for p in tmp_path.iterdir() if p.name != ".pruned")
343
+ # x.lock: created to take the orphan temp file's session lock before deleting it.
344
+ assert left == ["busy.json", "busy.lock", "idle.lock", "x.lock"]
345
+
346
+
347
+ def test_prune_leaves_a_temp_file_whose_session_is_locked(tmp_path):
348
+ import fcntl
349
+ import os
350
+ import time
351
+
352
+ from judgetap.guard.loop import SESSION_TTL_SECONDS, prune_sessions
353
+
354
+ old = time.time() - SESSION_TTL_SECONDS - 100
355
+ tmp = tmp_path / (
356
+ "busy." + "0123abcd" * 4 + ".tmp"
357
+ ) # <id>.<uuid hex>.tmp, as _save names it
358
+ lock = tmp_path / "busy.lock"
359
+ for p in (tmp, lock):
360
+ p.write_text("x")
361
+ os.utime(p, (old, old))
362
+ fd = os.open(lock, os.O_WRONLY)
363
+ fcntl.flock(fd, fcntl.LOCK_EX)
364
+ try:
365
+ prune_sessions(tmp_path)
366
+ finally:
367
+ os.close(fd)
368
+ assert tmp.exists()
369
+ prune_sessions(tmp_path, now=time.time() + 7200) # next sweep, lock free
370
+ assert not tmp.exists()
371
+
372
+
373
+ def test_lock_path_matches_session_lock_for_dotted_ids(tmp_path):
374
+ import uuid
375
+
376
+ from judgetap.guard.loop import _lock_for
377
+
378
+ d = tmp_path
379
+ for data in ("a.b.json", "a.b.stop"):
380
+ assert _lock_for(d / data) == (d / data).with_suffix(".lock") == d / "a.b.lock"
381
+ tmp = (d / data).with_suffix(f".{uuid.uuid4().hex}.tmp") # as _save writes it
382
+ assert _lock_for(tmp) == d / "a.b.lock"
@@ -212,3 +212,40 @@ def test_failed_judge_logs_its_calls_and_lets_the_agent_stop(tmp_path, monkeypat
212
212
  is None
213
213
  )
214
214
  assert load(tmp_path)["summary"]["engines"]["low"]["calls"] == 1
215
+
216
+
217
+ def test_stop_uses_last_assistant_message_when_the_transcript_lags(
218
+ tmp_path, monkeypatch
219
+ ):
220
+ import json
221
+
222
+ from judgetap.guard import stop
223
+ from judgetap.testing import StaticEngine
224
+
225
+ monkeypatch.setenv("JUDGETAP_HOME", str(tmp_path))
226
+ t = tmp_path / "t.jsonl"
227
+ t.write_text(
228
+ json.dumps({"type": "user", "message": {"content": "Fix all 3 failing tests"}})
229
+ + "\n"
230
+ )
231
+ seen = []
232
+
233
+ def judge(q, ctx):
234
+ seen.append(ctx["assistant_last_message"])
235
+ return {"yes": 0.05, "no": 0.95}
236
+
237
+ # The documented Stop input: the final reply comes in last_assistant_message.
238
+ payload = {
239
+ "session_id": "abc123",
240
+ "transcript_path": str(t),
241
+ "cwd": "/tmp",
242
+ "permission_mode": "default",
243
+ "hook_event_name": "Stop",
244
+ "stop_hook_active": False,
245
+ "last_assistant_message": "I fixed one test; two remain, stopping.",
246
+ "background_tasks": [],
247
+ "session_crons": [],
248
+ }
249
+ out = stop.handle(payload, engine=StaticEngine(judge, name="j"))
250
+ assert seen == ["I fixed one test; two remain, stopping."]
251
+ assert out["decision"] == "block"
@@ -0,0 +1,177 @@
1
+ import json
2
+ import math
3
+ import threading
4
+ import urllib.request
5
+
6
+ import pytest
7
+
8
+ from judgetap.dashboard.data import load
9
+
10
+
11
+ def write(tmp_path, rows):
12
+ (tmp_path / "guard.jsonl").write_text("\n".join(rows) + "\n")
13
+
14
+
15
+ @pytest.mark.parametrize(
16
+ "line",
17
+ [
18
+ '{"outcome":"allow","cost_usd":"0.1"}',
19
+ '{"outcome":["hold"]}',
20
+ '{"outcome":"allow","cost_usd":NaN}',
21
+ '{"outcome":"allow","layer":"judge","calls":[{"engine":"e","latency_ms":NaN}]}',
22
+ '{"outcome":"allow","layer":{"x":1},"engine":7,"source":[1],"p":"nope","calls":"x"}',
23
+ '{"outcome":"allow","latency_ms":Infinity,"layer":"judge","engine":"e"}',
24
+ ],
25
+ )
26
+ def test_odd_log_values_never_break_load(tmp_path, line):
27
+ write(tmp_path, [json.dumps({"outcome": "hold", "layer": "rules"}), line])
28
+ data = load(tmp_path)
29
+ json.dumps(data, allow_nan=False) # strict JSON: no NaN/Infinity left
30
+ assert data["summary"]["outcomes"].get("hold") == 1
31
+ assert all(isinstance(k, str) or k is None for k in data["summary"]["outcomes"])
32
+
33
+
34
+ def test_nan_latency_is_not_sampled(tmp_path):
35
+ write(
36
+ tmp_path,
37
+ [
38
+ '{"outcome":"allow","layer":"judge","calls":[{"engine":"e","latency_ms":NaN},{"engine":"e","latency_ms":5}]}'
39
+ ],
40
+ )
41
+ e = load(tmp_path)["summary"]["engines"]["e"]
42
+ assert e["calls"] == 2 and e["p50_ms"] == 5 and not math.isnan(e["p50_ms"])
43
+
44
+
45
+ def test_api_data_answers_with_a_bad_line(tmp_path):
46
+ from judgetap.dashboard.server import serve
47
+
48
+ write(tmp_path, ['{"outcome":"allow","cost_usd":"0.1"}', '{"outcome":["hold"]}'])
49
+ srv = serve(tmp_path, 0)
50
+ port = srv.server_address[1]
51
+ threading.Thread(target=srv.serve_forever, daemon=True).start()
52
+ try:
53
+ with urllib.request.urlopen(f"http://127.0.0.1:{port}/api/data") as res:
54
+ assert res.status == 200 and json.loads(res.read())["summary"]["total"] == 2
55
+ finally:
56
+ srv.shutdown()
57
+ srv.server_close()
58
+
59
+
60
+ def test_port_zero_uses_the_bound_port(tmp_path):
61
+ from judgetap.dashboard.server import serve
62
+
63
+ srv = serve(tmp_path, 0)
64
+ port = srv.server_address[1]
65
+ assert port != 0
66
+ threading.Thread(target=srv.serve_forever, daemon=True).start()
67
+ try:
68
+ with urllib.request.urlopen(f"http://127.0.0.1:{port}/") as res:
69
+ assert res.status == 200
70
+ finally:
71
+ srv.shutdown()
72
+ srv.server_close()
73
+
74
+
75
+ def test_keys_set_rejects_names_no_engine_reads(monkeypatch, capsys):
76
+ from judgetap.cli import main
77
+
78
+ asked = []
79
+ monkeypatch.setattr("getpass.getpass", lambda prompt: asked.append(prompt) or "v")
80
+ assert main(["keys", "set", "OPENAI_API_KEY"]) == 2
81
+ assert not asked and "isn't read by any engine" in capsys.readouterr().out
82
+ assert main(["keys", "status", "OPENAI_API_KEY"]) == 2
83
+
84
+
85
+ def test_empty_new_env_wins_over_old(monkeypatch):
86
+ from judgetap._compat import env, env_source
87
+
88
+ monkeypatch.setenv("JUDGETAP_ENGINE", "")
89
+ monkeypatch.setenv("SNAPJUDGE_ENGINE", "llm")
90
+ assert env("ENGINE") is None and env_source("ENGINE") is None
91
+ monkeypatch.delenv("JUDGETAP_ENGINE")
92
+ assert env("ENGINE") == "llm" and env_source("ENGINE") == "$SNAPJUDGE_ENGINE"
93
+
94
+
95
+ def test_nan_in_unknown_fields_is_scrubbed(tmp_path):
96
+ write(tmp_path, ['{"outcome":"allow","extra":{"x":[NaN, -Infinity]}}'])
97
+ json.dumps(load(tmp_path), allow_nan=False)
98
+
99
+
100
+ def test_huge_int_cost_keeps_the_record(tmp_path):
101
+ import json
102
+
103
+ from judgetap.dashboard.data import load
104
+
105
+ (tmp_path / "guard.jsonl").write_text(
106
+ json.dumps({"outcome": "hold", "cost_usd": 10**400}) + "\n"
107
+ )
108
+ s = load(tmp_path)["summary"]
109
+ assert s["total"] == 1 and s["outcomes"] == {"hold": 1}
110
+
111
+
112
+ def test_invalid_legacy_latency_adds_no_sample(tmp_path):
113
+ from judgetap.dashboard.data import load
114
+
115
+ (tmp_path / "guard.jsonl").write_text(
116
+ '{"outcome":"allow","layer":"judge","engine":"e","latency_ms":NaN}\n'
117
+ )
118
+ e = load(tmp_path)["summary"]["engines"]["e"]
119
+ assert e["calls"] == 1 and e["p50_ms"] is None
120
+
121
+
122
+ def test_overflowing_cost_total_still_serves(tmp_path):
123
+ import json
124
+ import socket
125
+ import threading
126
+ import urllib.request
127
+ from http.server import ThreadingHTTPServer
128
+
129
+ from judgetap.dashboard.server import make_handler
130
+
131
+ (tmp_path / "guard.jsonl").write_text(
132
+ "\n".join(json.dumps({"outcome": "allow", "cost_usd": 1e308}) for _ in range(2))
133
+ + "\n"
134
+ )
135
+ with socket.socket() as sock:
136
+ sock.bind(("127.0.0.1", 0))
137
+ port = sock.getsockname()[1]
138
+ srv = ThreadingHTTPServer(("127.0.0.1", port), make_handler(tmp_path, "tok", port))
139
+ threading.Thread(target=srv.serve_forever, daemon=True).start()
140
+ try:
141
+ with urllib.request.urlopen(f"http://127.0.0.1:{port}/api/data") as res:
142
+ assert res.status == 200
143
+ assert json.loads(res.read())["summary"]["cost_usd"] is None
144
+ finally:
145
+ srv.shutdown()
146
+ srv.server_close()
147
+
148
+
149
+ def test_malformed_calls_field_keeps_legacy_fallback(tmp_path):
150
+ import json
151
+
152
+ from judgetap.dashboard.data import load
153
+
154
+ (tmp_path / "guard.jsonl").write_text(
155
+ json.dumps(
156
+ {
157
+ "id": "a",
158
+ "outcome": "allow",
159
+ "layer": "judge",
160
+ "engine": "e",
161
+ "latency_ms": 12,
162
+ "calls": "x",
163
+ }
164
+ )
165
+ + "\n"
166
+ )
167
+ e = load(tmp_path)["summary"]["engines"]["e"]
168
+ assert e["calls"] == 1 and e["p50_ms"] == 12
169
+
170
+
171
+ def test_deeply_nested_extra_field_does_not_drop_the_record(tmp_path):
172
+ from judgetap.dashboard.data import load
173
+
174
+ deep = "[" * 5000 + "]" * 5000
175
+ line = '{"id": "a", "outcome": "hold", "layer": "rules", "extra": ' + deep + "}"
176
+ (tmp_path / "guard.jsonl").write_text(line + "\n")
177
+ assert load(tmp_path)["summary"]["total"] == 1
File without changes
File without changes