judgetap 0.1.1.dev32__tar.gz → 0.1.1.dev34__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/PKG-INFO +2 -2
  2. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/README.md +1 -1
  3. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/pyproject.toml +1 -1
  4. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/__init__.py +1 -1
  5. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/guard/core.py +20 -0
  6. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/guard/hook.py +30 -21
  7. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/guard/rules.py +2 -0
  8. judgetap-0.1.1.dev34/tests/test_guard_polish_76.py +139 -0
  9. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/.github/workflows/ci.yml +0 -0
  10. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/.github/workflows/demo.yml +0 -0
  11. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/.github/workflows/release.yml +0 -0
  12. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/.gitignore +0 -0
  13. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/.python-version +0 -0
  14. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/.release-please-manifest.json +0 -0
  15. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/CHANGELOG.md +0 -0
  16. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/CONTRIBUTING.md +0 -0
  17. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/LICENSE +0 -0
  18. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/docs/SPEC.md +0 -0
  19. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/docs/demo.tape +0 -0
  20. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/release-please-config.json +0 -0
  21. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/_compat.py +0 -0
  22. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/api.py +0 -0
  23. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/cascade.py +0 -0
  24. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/cli.py +0 -0
  25. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/dashboard/__init__.py +0 -0
  26. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/dashboard/data.py +0 -0
  27. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/dashboard/page.html +0 -0
  28. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/dashboard/server.py +0 -0
  29. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/decision_log.py +0 -0
  30. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/engine.py +0 -0
  31. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/engines/__init__.py +0 -0
  32. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/engines/agentjev.py +0 -0
  33. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/engines/jev.py +0 -0
  34. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/engines/laya.py +0 -0
  35. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/engines/llm.py +0 -0
  36. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/errors.py +0 -0
  37. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/evaluate.py +0 -0
  38. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/guard/__init__.py +0 -0
  39. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/guard/install.py +0 -0
  40. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/guard/loop.py +0 -0
  41. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/guard/stop.py +0 -0
  42. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/py.typed +0 -0
  43. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/secrets.py +0 -0
  44. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/testing.py +0 -0
  45. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/types.py +0 -0
  46. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_api.py +0 -0
  47. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_call_accounting.py +0 -0
  48. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_calls.py +0 -0
  49. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_cascade.py +0 -0
  50. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_dashboard.py +0 -0
  51. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_decision_log.py +0 -0
  52. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_engine_jev.py +0 -0
  53. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_engine_llm.py +0 -0
  54. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_engine_local.py +0 -0
  55. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_evaluate.py +0 -0
  56. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_guard_agents.py +0 -0
  57. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_guard_core.py +0 -0
  58. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_guard_hook.py +0 -0
  59. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_guard_loop.py +0 -0
  60. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_guard_rules.py +0 -0
  61. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_guard_stop.py +0 -0
  62. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_guard_trust.py +0 -0
  63. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_questions.py +0 -0
  64. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_robustness_68.py +0 -0
  65. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_secrets.py +0 -0
  66. {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/uv.lock +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: judgetap
3
- Version: 0.1.1.dev32
3
+ Version: 0.1.1.dev34
4
4
  Summary: Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development.
5
5
  Project-URL: Homepage, https://github.com/mergesafe-ai/judgetap
6
6
  Author-email: Omer Bar-Ness <omer@zsquared.io>
@@ -42,7 +42,7 @@ judgetap guard test "git push --force origin main"
42
42
 
43
43
  ## What it does
44
44
 
45
- - **Guard.** A hook that checks every shell command (and, in Claude Code, every file write and edit) before it runs. Rules catch common destructive forms: recursive deletes outside the workspace, force-pushes and pushes to protected branches, `DROP`/`DELETE` without `WHERE`, `terraform destroy`, and secrets written to files. With an engine configured, a model judges the rest: is it irreversible? off-task? against a rule in `AGENTS.md`? Without an engine, the rules fail closed for shell commands (anything they can't vouch for asks you); file writes and edits are only checked for secrets and your own rules.
45
+ - **Guard.** A hook that checks every shell command (and, in Claude Code, every file write and edit) before it runs. Rules catch common destructive forms: recursive deletes outside the workspace, force-pushes and pushes to protected branches, `DROP`/`DELETE` without `WHERE`, `terraform destroy`, and secrets written to files. With an engine configured, a model judges the rest: is it irreversible? off-task? against a rule in `AGENTS.md`? Without an engine, the rules fail closed for shell commands (anything they can't vouch for asks you); file writes and edits are checked for secrets, your own rules, and writes to the guard's own configuration (which ask).
46
46
  - **Library.** One API (`choice`, `score`, `yesno`, `batch`) over every Jev-style decision engine, with a cascade that escalates low-confidence answers to a stronger engine.
47
47
  - **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost.
48
48
  - **Dashboard.** `judgetap dashboard` is a local page with recent decisions, holds, asks, latency and cost per engine. You can mark a hold as a false alarm.
@@ -25,7 +25,7 @@ judgetap guard test "git push --force origin main"
25
25
 
26
26
  ## What it does
27
27
 
28
- - **Guard.** A hook that checks every shell command (and, in Claude Code, every file write and edit) before it runs. Rules catch common destructive forms: recursive deletes outside the workspace, force-pushes and pushes to protected branches, `DROP`/`DELETE` without `WHERE`, `terraform destroy`, and secrets written to files. With an engine configured, a model judges the rest: is it irreversible? off-task? against a rule in `AGENTS.md`? Without an engine, the rules fail closed for shell commands (anything they can't vouch for asks you); file writes and edits are only checked for secrets and your own rules.
28
+ - **Guard.** A hook that checks every shell command (and, in Claude Code, every file write and edit) before it runs. Rules catch common destructive forms: recursive deletes outside the workspace, force-pushes and pushes to protected branches, `DROP`/`DELETE` without `WHERE`, `terraform destroy`, and secrets written to files. With an engine configured, a model judges the rest: is it irreversible? off-task? against a rule in `AGENTS.md`? Without an engine, the rules fail closed for shell commands (anything they can't vouch for asks you); file writes and edits are checked for secrets, your own rules, and writes to the guard's own configuration (which ask).
29
29
  - **Library.** One API (`choice`, `score`, `yesno`, `batch`) over every Jev-style decision engine, with a cascade that escalates low-confidence answers to a stronger engine.
30
30
  - **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost.
31
31
  - **Dashboard.** `judgetap dashboard` is a local page with recent decisions, holds, asks, latency and cost per engine. You can mark a hold as a false alarm.
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "judgetap"
3
- version = "0.1.1.dev32"
3
+ version = "0.1.1.dev34"
4
4
  description = "Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development."
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -23,7 +23,7 @@ from judgetap.errors import (
23
23
  )
24
24
  from judgetap.types import Decision, Question
25
25
 
26
- __version__ = "0.1.1.dev32" # x-release-please-version
26
+ __version__ = "0.1.1.dev34" # x-release-please-version
27
27
 
28
28
  __all__ = [
29
29
  "Cascade",
@@ -99,6 +99,10 @@ def load_user_rules(path: Path, *, trusted: bool = True) -> list[UserRule]:
99
99
  return []
100
100
  with path.open("rb") as fh:
101
101
  data = tomllib.load(fh)
102
+ return _rules_from(data, path, trusted)
103
+
104
+
105
+ def _rules_from(data: dict, path: Path, trusted: bool) -> list[UserRule]:
102
106
  rules = []
103
107
  for entry in data.get("rule", []):
104
108
  outcome = entry.get("outcome", "hold")
@@ -229,6 +233,22 @@ def project_rules(cwd: Path) -> str | None:
229
233
  return None
230
234
 
231
235
 
236
+ def load_rules_counting_allows(
237
+ path: Path, *, trusted: bool = True
238
+ ) -> tuple[list[UserRule], int]:
239
+ """load_user_rules plus how many `allow` rules were dropped, from one
240
+ parse of the file (the hook calls this on every action)."""
241
+ if not path.is_file():
242
+ return [], 0
243
+ with path.open("rb") as fh:
244
+ data = tomllib.load(fh)
245
+ return _rules_from(data, path, trusted), (
246
+ 0
247
+ if trusted
248
+ else sum(1 for e in data.get("rule", []) if e.get("outcome", "hold") == "allow")
249
+ )
250
+
251
+
232
252
  def dropped_allows(path: Path) -> int:
233
253
  """How many `allow` rules a repo guard.toml has (ignored when loaded untrusted)."""
234
254
  if not path.is_file():
@@ -24,8 +24,7 @@ from judgetap.guard.core import (
24
24
  Action,
25
25
  Verdict,
26
26
  check,
27
- dropped_allows,
28
- load_user_rules,
27
+ load_rules_counting_allows,
29
28
  project_rules,
30
29
  )
31
30
  from judgetap.guard.rules import redact
@@ -56,11 +55,15 @@ def action_from_hook(payload: dict[str, Any]) -> Action | None:
56
55
  action = Action(tool=tool, cwd=cwd)
57
56
  if tool == "Bash":
58
57
  command = inp.get("command")
59
- # None means the command couldn't be read: _decide asks rather than allows.
60
- action.command = None if command is None else _text(command)
61
- action.unreadable = command is None
58
+ # Only a nonempty string can be inspected; anything else asks.
59
+ readable = isinstance(command, str) and bool(command.strip())
60
+ action.command = command if readable else None
61
+ action.unreadable = not readable
62
62
  else:
63
- action.path = _text(inp.get("file_path")) or None
63
+ path = inp.get("file_path")
64
+ action.path = path if isinstance(path, str) and path.strip() else None
65
+ # A write whose destination can't be read can't be checked: ask.
66
+ action.unreadable = action.path is None
64
67
  if tool == "Write":
65
68
  action.content = _text(inp.get("content"))
66
69
  elif tool == "Edit":
@@ -214,17 +217,21 @@ def normalise(payload: dict[str, Any], agent: str) -> dict[str, Any]:
214
217
  if agent == "cursor": # beforeShellExecution: {command, cwd, ...}
215
218
  return {
216
219
  "tool_name": "Bash",
217
- "tool_input": {"command": payload.get("command", "")},
220
+ # Missing stays None, so the guard asks instead of checking "".
221
+ "tool_input": {"command": payload.get("command")},
218
222
  "cwd": payload.get("cwd"),
219
223
  "session_id": payload.get("conversation_id"),
220
224
  "transcript_path": payload.get("transcript_path"),
221
225
  }
222
226
  if agent == "codex" and payload.get("tool_name") in CODEX_SHELL_TOOLS:
223
227
  # Codex's shell tool is exec_command with tool_input.cmd (str or argv).
224
- inp = payload.get("tool_input") or {}
225
- cmd = inp.get("cmd", inp.get("command", ""))
226
- if isinstance(cmd, list):
228
+ inp = payload.get("tool_input")
229
+ inp = inp if isinstance(inp, dict) else {}
230
+ cmd = inp.get("cmd", inp.get("command"))
231
+ if isinstance(cmd, list) and cmd:
227
232
  cmd = shlex.join(str(c) for c in cmd)
233
+ elif not isinstance(cmd, str):
234
+ cmd = None # missing or not a command: unreadable, so the guard asks
228
235
  return {**payload, "tool_name": "Bash", "tool_input": {"command": cmd}}
229
236
  return payload # Claude Code's PreToolUse shape
230
237
 
@@ -261,6 +268,8 @@ def respond(
261
268
  reason = _reason(verdict)
262
269
  if verdict.outcome == "ask":
263
270
  reason += " Ask the user before running it."
271
+ if note:
272
+ reason += f" ({verdict.error})" # warnings aren't lost on a deny
264
273
  return None, 2, reason
265
274
  return {}, 0, note
266
275
  out = {}
@@ -336,11 +345,12 @@ def _warn_once(session: str | None, key: str) -> bool:
336
345
 
337
346
  def _decide(action: Action, start: float, session: str | None = None) -> Verdict:
338
347
  if getattr(action, "unreadable", False):
339
- # A Bash call whose command can't be read: fail closed.
348
+ # A command or destination that can't be read: fail closed.
349
+ what = "command" if action.tool == "Bash" else "destination file"
340
350
  return Verdict(
341
351
  "ask",
342
352
  "rules",
343
- "the command couldn't be read from the hook input",
353
+ f"the {what} couldn't be read from the hook input",
344
354
  rule="unreadable",
345
355
  )
346
356
  try:
@@ -355,19 +365,18 @@ def _decide(action: Action, start: float, session: str | None = None) -> Verdict
355
365
  (action.cwd / "guard.toml", False),
356
366
  ):
357
367
  try:
358
- rules += load_user_rules(path, trusted=trusted)
359
- if (
360
- not trusted
361
- and (n := dropped_allows(path))
362
- and _warn_once(session, f"repo-allow:{path}")
363
- ):
368
+ loaded, n = load_rules_counting_allows(path, trusted=trusted)
369
+ rules += loaded
370
+ if n and _warn_once(session, f"repo-allow:{path}"):
364
371
  config_error = f"ignored {n} allow rule(s) in {path}: a repo can only tighten the guard"
365
372
  except Exception as err: # noqa: BLE001 -- a bad config must not disable built-ins
366
373
  config_error = f"ignored {path}: {err}"
367
374
  verdict = check(action, engine, rules)
368
- problem = engine_error or config_error
369
- if problem and not verdict.error:
370
- verdict.error = problem
375
+ # Keep every problem: an engine error must not hide a config warning that
376
+ # _warn_once has already marked as shown for this session.
377
+ problems = [p for p in (verdict.error, engine_error, config_error) if p]
378
+ if problems:
379
+ verdict.error = "; ".join(problems)
371
380
  verdict.reason = verdict.reason or "engine unavailable; rules only"
372
381
  verdict.latency_ms = verdict.latency_ms or (time.perf_counter() - start) * 1000
373
382
  return verdict
@@ -922,6 +922,8 @@ def check_path(
922
922
  GUARD_CONFIG_MENTION = re.compile(
923
923
  r"(?:guard\.toml|\.claude/settings[^\s'\"/]*\.json|\.cursor/hooks\.json"
924
924
  r"|\.codex/hooks\.json|\.codex/config\.toml)"
925
+ # A whole name: guard.toml.example or settings.json.bak is another file.
926
+ r"(?![\w.-])"
925
927
  )
926
928
 
927
929
 
@@ -0,0 +1,139 @@
1
+ """Fixes for the P2s left on #71 (issue #76)."""
2
+
3
+ import io
4
+ import json
5
+
6
+ import pytest
7
+
8
+ from judgetap.guard import hook
9
+
10
+
11
+ @pytest.fixture(autouse=True)
12
+ def _home(tmp_path, monkeypatch):
13
+ monkeypatch.setenv("JUDGETAP_HOME", str(tmp_path / "home"))
14
+ monkeypatch.delenv("JUDGETAP_ENGINE", raising=False)
15
+ monkeypatch.delenv("SNAPJUDGE_ENGINE", raising=False)
16
+
17
+
18
+ def run(payload, agent="claude-code"):
19
+ out, err = io.StringIO(), io.StringIO()
20
+ code = hook.run(io.StringIO(json.dumps(payload)), out, err, agent=agent)
21
+ return (
22
+ code,
23
+ (json.loads(out.getvalue()) if out.getvalue() else None),
24
+ err.getvalue(),
25
+ )
26
+
27
+
28
+ # P2 1, 2 and 5: missing or malformed commands from Cursor and Codex ask.
29
+ def test_cursor_missing_command_asks(tmp_path):
30
+ _, out, _ = run({"cwd": str(tmp_path)}, agent="cursor")
31
+ assert out["permission"] == "ask"
32
+
33
+
34
+ @pytest.mark.parametrize(
35
+ "tool_input",
36
+ [{}, [], "ls", {"cmd": 5}, {"cmd": []}],
37
+ )
38
+ def test_codex_malformed_tool_input_asks(tmp_path, tool_input):
39
+ code, _, err = run(
40
+ {"tool_name": "exec_command", "tool_input": tool_input, "cwd": str(tmp_path)},
41
+ agent="codex",
42
+ )
43
+ assert code == 2 and "couldn't be read" in err
44
+
45
+
46
+ # P2 7: a non-string or blank Bash command isn't readable.
47
+ @pytest.mark.parametrize("command", [[], {"a": 1}, 5, "", " "])
48
+ def test_non_string_command_asks(tmp_path, command):
49
+ _, out, _ = run(
50
+ {"tool_name": "Bash", "tool_input": {"command": command}, "cwd": str(tmp_path)}
51
+ )
52
+ assert out["hookSpecificOutput"]["permissionDecision"] == "ask"
53
+
54
+
55
+ # P2 8: a write without a readable destination asks.
56
+ @pytest.mark.parametrize("tool", ["Write", "Edit", "MultiEdit"])
57
+ def test_write_without_destination_asks(tmp_path, tool):
58
+ _, out, _ = run(
59
+ {
60
+ "tool_name": tool,
61
+ "tool_input": {"content": "ordinary text"},
62
+ "cwd": str(tmp_path),
63
+ }
64
+ )
65
+ assert out["hookSpecificOutput"]["permissionDecision"] == "ask"
66
+ assert "destination" in out["hookSpecificOutput"]["permissionDecisionReason"]
67
+
68
+
69
+ # P2 3: example and backup files aren't the active configuration.
70
+ @pytest.mark.parametrize(
71
+ ("command", "asks"),
72
+ [
73
+ ("cat guard.toml.example", False),
74
+ ("cp .claude/settings.json.backup /tmp/x", False),
75
+ ("cat guard.toml", True),
76
+ ("echo x >> .claude/settings.local.json", True),
77
+ ],
78
+ )
79
+ def test_mention_needs_the_whole_config_name(tmp_path, command, asks):
80
+ from judgetap.guard.rules import check_command_writes
81
+
82
+ assert (check_command_writes(command, tmp_path) is not None) is asks
83
+
84
+
85
+ # P2 4: the repo config is parsed once per action.
86
+ def test_repo_config_parsed_once(tmp_path, monkeypatch):
87
+ import tomllib
88
+
89
+ (tmp_path / "guard.toml").write_text('[[rule]]\npattern = "x"\noutcome = "allow"\n')
90
+ calls = []
91
+ real = tomllib.load
92
+ monkeypatch.setattr(tomllib, "load", lambda fh: calls.append(1) or real(fh))
93
+ run(
94
+ {
95
+ "tool_name": "Bash",
96
+ "tool_input": {"command": "ls"},
97
+ "cwd": str(tmp_path),
98
+ "session_id": "s1",
99
+ }
100
+ )
101
+ assert len(calls) == 1 # ~/.judgetap/guard.toml doesn't exist here
102
+
103
+
104
+ # P2 6: the ignored-allow warning survives an engine error and a Codex deny.
105
+ def test_repo_warning_not_lost_behind_engine_error(tmp_path, monkeypatch):
106
+ monkeypatch.setenv("JUDGETAP_ENGINE", "nonsense")
107
+ (tmp_path / "guard.toml").write_text('[[rule]]\npattern = "x"\noutcome = "allow"\n')
108
+ _, out, _ = run(
109
+ {
110
+ "tool_name": "Bash",
111
+ "tool_input": {"command": "make x"},
112
+ "cwd": str(tmp_path),
113
+ "session_id": "s2",
114
+ }
115
+ )
116
+ msg = out["systemMessage"]
117
+ assert "ignored 1 allow rule" in msg and "nonsense" in msg
118
+
119
+
120
+ def test_repo_warning_shown_on_codex_deny(tmp_path):
121
+ (tmp_path / "guard.toml").write_text('[[rule]]\npattern = "x"\noutcome = "allow"\n')
122
+ code, _, err = run(
123
+ {
124
+ "tool_name": "exec_command",
125
+ "tool_input": {"cmd": "git push --force"},
126
+ "cwd": str(tmp_path),
127
+ "session_id": "s3",
128
+ },
129
+ agent="codex",
130
+ )
131
+ assert code == 2 and "ignored 1 allow rule" in err
132
+
133
+
134
+ # P2 9: README describes the guard-config check.
135
+ def test_readme_mentions_guard_config_check():
136
+ from pathlib import Path
137
+
138
+ readme = (Path(__file__).parent.parent / "README.md").read_text()
139
+ assert "writes to the guard's own configuration" in readme
File without changes
File without changes