judgetap 0.1.1.dev32__tar.gz → 0.1.1.dev34__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/PKG-INFO +2 -2
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/README.md +1 -1
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/pyproject.toml +1 -1
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/__init__.py +1 -1
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/guard/core.py +20 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/guard/hook.py +30 -21
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/guard/rules.py +2 -0
- judgetap-0.1.1.dev34/tests/test_guard_polish_76.py +139 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/.github/workflows/ci.yml +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/.github/workflows/demo.yml +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/.github/workflows/release.yml +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/.gitignore +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/.python-version +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/.release-please-manifest.json +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/CHANGELOG.md +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/CONTRIBUTING.md +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/LICENSE +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/docs/SPEC.md +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/docs/demo.tape +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/release-please-config.json +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/_compat.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/api.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/cascade.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/cli.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/dashboard/__init__.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/dashboard/data.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/dashboard/page.html +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/dashboard/server.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/decision_log.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/engine.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/engines/__init__.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/engines/agentjev.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/engines/jev.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/engines/laya.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/engines/llm.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/errors.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/evaluate.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/guard/__init__.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/guard/install.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/guard/loop.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/guard/stop.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/py.typed +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/secrets.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/testing.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/src/judgetap/types.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_api.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_call_accounting.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_calls.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_cascade.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_dashboard.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_decision_log.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_engine_jev.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_engine_llm.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_engine_local.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_evaluate.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_guard_agents.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_guard_core.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_guard_hook.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_guard_loop.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_guard_rules.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_guard_stop.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_guard_trust.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_questions.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_robustness_68.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/tests/test_secrets.py +0 -0
- {judgetap-0.1.1.dev32 → judgetap-0.1.1.dev34}/uv.lock +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: judgetap
|
|
3
|
-
Version: 0.1.1.
|
|
3
|
+
Version: 0.1.1.dev34
|
|
4
4
|
Summary: Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development.
|
|
5
5
|
Project-URL: Homepage, https://github.com/mergesafe-ai/judgetap
|
|
6
6
|
Author-email: Omer Bar-Ness <omer@zsquared.io>
|
|
@@ -42,7 +42,7 @@ judgetap guard test "git push --force origin main"
|
|
|
42
42
|
|
|
43
43
|
## What it does
|
|
44
44
|
|
|
45
|
-
- **Guard.** A hook that checks every shell command (and, in Claude Code, every file write and edit) before it runs. Rules catch common destructive forms: recursive deletes outside the workspace, force-pushes and pushes to protected branches, `DROP`/`DELETE` without `WHERE`, `terraform destroy`, and secrets written to files. With an engine configured, a model judges the rest: is it irreversible? off-task? against a rule in `AGENTS.md`? Without an engine, the rules fail closed for shell commands (anything they can't vouch for asks you); file writes and edits are
|
|
45
|
+
- **Guard.** A hook that checks every shell command (and, in Claude Code, every file write and edit) before it runs. Rules catch common destructive forms: recursive deletes outside the workspace, force-pushes and pushes to protected branches, `DROP`/`DELETE` without `WHERE`, `terraform destroy`, and secrets written to files. With an engine configured, a model judges the rest: is it irreversible? off-task? against a rule in `AGENTS.md`? Without an engine, the rules fail closed for shell commands (anything they can't vouch for asks you); file writes and edits are checked for secrets, your own rules, and writes to the guard's own configuration (which ask).
|
|
46
46
|
- **Library.** One API (`choice`, `score`, `yesno`, `batch`) over every Jev-style decision engine, with a cascade that escalates low-confidence answers to a stronger engine.
|
|
47
47
|
- **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost.
|
|
48
48
|
- **Dashboard.** `judgetap dashboard` is a local page with recent decisions, holds, asks, latency and cost per engine. You can mark a hold as a false alarm.
|
|
@@ -25,7 +25,7 @@ judgetap guard test "git push --force origin main"
|
|
|
25
25
|
|
|
26
26
|
## What it does
|
|
27
27
|
|
|
28
|
-
- **Guard.** A hook that checks every shell command (and, in Claude Code, every file write and edit) before it runs. Rules catch common destructive forms: recursive deletes outside the workspace, force-pushes and pushes to protected branches, `DROP`/`DELETE` without `WHERE`, `terraform destroy`, and secrets written to files. With an engine configured, a model judges the rest: is it irreversible? off-task? against a rule in `AGENTS.md`? Without an engine, the rules fail closed for shell commands (anything they can't vouch for asks you); file writes and edits are
|
|
28
|
+
- **Guard.** A hook that checks every shell command (and, in Claude Code, every file write and edit) before it runs. Rules catch common destructive forms: recursive deletes outside the workspace, force-pushes and pushes to protected branches, `DROP`/`DELETE` without `WHERE`, `terraform destroy`, and secrets written to files. With an engine configured, a model judges the rest: is it irreversible? off-task? against a rule in `AGENTS.md`? Without an engine, the rules fail closed for shell commands (anything they can't vouch for asks you); file writes and edits are checked for secrets, your own rules, and writes to the guard's own configuration (which ask).
|
|
29
29
|
- **Library.** One API (`choice`, `score`, `yesno`, `batch`) over every Jev-style decision engine, with a cascade that escalates low-confidence answers to a stronger engine.
|
|
30
30
|
- **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost.
|
|
31
31
|
- **Dashboard.** `judgetap dashboard` is a local page with recent decisions, holds, asks, latency and cost per engine. You can mark a hold as a false alarm.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "judgetap"
|
|
3
|
-
version = "0.1.1.
|
|
3
|
+
version = "0.1.1.dev34"
|
|
4
4
|
description = "Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "Apache-2.0"
|
|
@@ -99,6 +99,10 @@ def load_user_rules(path: Path, *, trusted: bool = True) -> list[UserRule]:
|
|
|
99
99
|
return []
|
|
100
100
|
with path.open("rb") as fh:
|
|
101
101
|
data = tomllib.load(fh)
|
|
102
|
+
return _rules_from(data, path, trusted)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _rules_from(data: dict, path: Path, trusted: bool) -> list[UserRule]:
|
|
102
106
|
rules = []
|
|
103
107
|
for entry in data.get("rule", []):
|
|
104
108
|
outcome = entry.get("outcome", "hold")
|
|
@@ -229,6 +233,22 @@ def project_rules(cwd: Path) -> str | None:
|
|
|
229
233
|
return None
|
|
230
234
|
|
|
231
235
|
|
|
236
|
+
def load_rules_counting_allows(
|
|
237
|
+
path: Path, *, trusted: bool = True
|
|
238
|
+
) -> tuple[list[UserRule], int]:
|
|
239
|
+
"""load_user_rules plus how many `allow` rules were dropped, from one
|
|
240
|
+
parse of the file (the hook calls this on every action)."""
|
|
241
|
+
if not path.is_file():
|
|
242
|
+
return [], 0
|
|
243
|
+
with path.open("rb") as fh:
|
|
244
|
+
data = tomllib.load(fh)
|
|
245
|
+
return _rules_from(data, path, trusted), (
|
|
246
|
+
0
|
|
247
|
+
if trusted
|
|
248
|
+
else sum(1 for e in data.get("rule", []) if e.get("outcome", "hold") == "allow")
|
|
249
|
+
)
|
|
250
|
+
|
|
251
|
+
|
|
232
252
|
def dropped_allows(path: Path) -> int:
|
|
233
253
|
"""How many `allow` rules a repo guard.toml has (ignored when loaded untrusted)."""
|
|
234
254
|
if not path.is_file():
|
|
@@ -24,8 +24,7 @@ from judgetap.guard.core import (
|
|
|
24
24
|
Action,
|
|
25
25
|
Verdict,
|
|
26
26
|
check,
|
|
27
|
-
|
|
28
|
-
load_user_rules,
|
|
27
|
+
load_rules_counting_allows,
|
|
29
28
|
project_rules,
|
|
30
29
|
)
|
|
31
30
|
from judgetap.guard.rules import redact
|
|
@@ -56,11 +55,15 @@ def action_from_hook(payload: dict[str, Any]) -> Action | None:
|
|
|
56
55
|
action = Action(tool=tool, cwd=cwd)
|
|
57
56
|
if tool == "Bash":
|
|
58
57
|
command = inp.get("command")
|
|
59
|
-
#
|
|
60
|
-
|
|
61
|
-
action.
|
|
58
|
+
# Only a nonempty string can be inspected; anything else asks.
|
|
59
|
+
readable = isinstance(command, str) and bool(command.strip())
|
|
60
|
+
action.command = command if readable else None
|
|
61
|
+
action.unreadable = not readable
|
|
62
62
|
else:
|
|
63
|
-
|
|
63
|
+
path = inp.get("file_path")
|
|
64
|
+
action.path = path if isinstance(path, str) and path.strip() else None
|
|
65
|
+
# A write whose destination can't be read can't be checked: ask.
|
|
66
|
+
action.unreadable = action.path is None
|
|
64
67
|
if tool == "Write":
|
|
65
68
|
action.content = _text(inp.get("content"))
|
|
66
69
|
elif tool == "Edit":
|
|
@@ -214,17 +217,21 @@ def normalise(payload: dict[str, Any], agent: str) -> dict[str, Any]:
|
|
|
214
217
|
if agent == "cursor": # beforeShellExecution: {command, cwd, ...}
|
|
215
218
|
return {
|
|
216
219
|
"tool_name": "Bash",
|
|
217
|
-
|
|
220
|
+
# Missing stays None, so the guard asks instead of checking "".
|
|
221
|
+
"tool_input": {"command": payload.get("command")},
|
|
218
222
|
"cwd": payload.get("cwd"),
|
|
219
223
|
"session_id": payload.get("conversation_id"),
|
|
220
224
|
"transcript_path": payload.get("transcript_path"),
|
|
221
225
|
}
|
|
222
226
|
if agent == "codex" and payload.get("tool_name") in CODEX_SHELL_TOOLS:
|
|
223
227
|
# Codex's shell tool is exec_command with tool_input.cmd (str or argv).
|
|
224
|
-
inp = payload.get("tool_input")
|
|
225
|
-
|
|
226
|
-
|
|
228
|
+
inp = payload.get("tool_input")
|
|
229
|
+
inp = inp if isinstance(inp, dict) else {}
|
|
230
|
+
cmd = inp.get("cmd", inp.get("command"))
|
|
231
|
+
if isinstance(cmd, list) and cmd:
|
|
227
232
|
cmd = shlex.join(str(c) for c in cmd)
|
|
233
|
+
elif not isinstance(cmd, str):
|
|
234
|
+
cmd = None # missing or not a command: unreadable, so the guard asks
|
|
228
235
|
return {**payload, "tool_name": "Bash", "tool_input": {"command": cmd}}
|
|
229
236
|
return payload # Claude Code's PreToolUse shape
|
|
230
237
|
|
|
@@ -261,6 +268,8 @@ def respond(
|
|
|
261
268
|
reason = _reason(verdict)
|
|
262
269
|
if verdict.outcome == "ask":
|
|
263
270
|
reason += " Ask the user before running it."
|
|
271
|
+
if note:
|
|
272
|
+
reason += f" ({verdict.error})" # warnings aren't lost on a deny
|
|
264
273
|
return None, 2, reason
|
|
265
274
|
return {}, 0, note
|
|
266
275
|
out = {}
|
|
@@ -336,11 +345,12 @@ def _warn_once(session: str | None, key: str) -> bool:
|
|
|
336
345
|
|
|
337
346
|
def _decide(action: Action, start: float, session: str | None = None) -> Verdict:
|
|
338
347
|
if getattr(action, "unreadable", False):
|
|
339
|
-
# A
|
|
348
|
+
# A command or destination that can't be read: fail closed.
|
|
349
|
+
what = "command" if action.tool == "Bash" else "destination file"
|
|
340
350
|
return Verdict(
|
|
341
351
|
"ask",
|
|
342
352
|
"rules",
|
|
343
|
-
"the
|
|
353
|
+
f"the {what} couldn't be read from the hook input",
|
|
344
354
|
rule="unreadable",
|
|
345
355
|
)
|
|
346
356
|
try:
|
|
@@ -355,19 +365,18 @@ def _decide(action: Action, start: float, session: str | None = None) -> Verdict
|
|
|
355
365
|
(action.cwd / "guard.toml", False),
|
|
356
366
|
):
|
|
357
367
|
try:
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
and (n := dropped_allows(path))
|
|
362
|
-
and _warn_once(session, f"repo-allow:{path}")
|
|
363
|
-
):
|
|
368
|
+
loaded, n = load_rules_counting_allows(path, trusted=trusted)
|
|
369
|
+
rules += loaded
|
|
370
|
+
if n and _warn_once(session, f"repo-allow:{path}"):
|
|
364
371
|
config_error = f"ignored {n} allow rule(s) in {path}: a repo can only tighten the guard"
|
|
365
372
|
except Exception as err: # noqa: BLE001 -- a bad config must not disable built-ins
|
|
366
373
|
config_error = f"ignored {path}: {err}"
|
|
367
374
|
verdict = check(action, engine, rules)
|
|
368
|
-
problem
|
|
369
|
-
|
|
370
|
-
|
|
375
|
+
# Keep every problem: an engine error must not hide a config warning that
|
|
376
|
+
# _warn_once has already marked as shown for this session.
|
|
377
|
+
problems = [p for p in (verdict.error, engine_error, config_error) if p]
|
|
378
|
+
if problems:
|
|
379
|
+
verdict.error = "; ".join(problems)
|
|
371
380
|
verdict.reason = verdict.reason or "engine unavailable; rules only"
|
|
372
381
|
verdict.latency_ms = verdict.latency_ms or (time.perf_counter() - start) * 1000
|
|
373
382
|
return verdict
|
|
@@ -922,6 +922,8 @@ def check_path(
|
|
|
922
922
|
GUARD_CONFIG_MENTION = re.compile(
|
|
923
923
|
r"(?:guard\.toml|\.claude/settings[^\s'\"/]*\.json|\.cursor/hooks\.json"
|
|
924
924
|
r"|\.codex/hooks\.json|\.codex/config\.toml)"
|
|
925
|
+
# A whole name: guard.toml.example or settings.json.bak is another file.
|
|
926
|
+
r"(?![\w.-])"
|
|
925
927
|
)
|
|
926
928
|
|
|
927
929
|
|
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
"""Fixes for the P2s left on #71 (issue #76)."""
|
|
2
|
+
|
|
3
|
+
import io
|
|
4
|
+
import json
|
|
5
|
+
|
|
6
|
+
import pytest
|
|
7
|
+
|
|
8
|
+
from judgetap.guard import hook
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@pytest.fixture(autouse=True)
|
|
12
|
+
def _home(tmp_path, monkeypatch):
|
|
13
|
+
monkeypatch.setenv("JUDGETAP_HOME", str(tmp_path / "home"))
|
|
14
|
+
monkeypatch.delenv("JUDGETAP_ENGINE", raising=False)
|
|
15
|
+
monkeypatch.delenv("SNAPJUDGE_ENGINE", raising=False)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def run(payload, agent="claude-code"):
|
|
19
|
+
out, err = io.StringIO(), io.StringIO()
|
|
20
|
+
code = hook.run(io.StringIO(json.dumps(payload)), out, err, agent=agent)
|
|
21
|
+
return (
|
|
22
|
+
code,
|
|
23
|
+
(json.loads(out.getvalue()) if out.getvalue() else None),
|
|
24
|
+
err.getvalue(),
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
# P2 1, 2 and 5: missing or malformed commands from Cursor and Codex ask.
|
|
29
|
+
def test_cursor_missing_command_asks(tmp_path):
|
|
30
|
+
_, out, _ = run({"cwd": str(tmp_path)}, agent="cursor")
|
|
31
|
+
assert out["permission"] == "ask"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@pytest.mark.parametrize(
|
|
35
|
+
"tool_input",
|
|
36
|
+
[{}, [], "ls", {"cmd": 5}, {"cmd": []}],
|
|
37
|
+
)
|
|
38
|
+
def test_codex_malformed_tool_input_asks(tmp_path, tool_input):
|
|
39
|
+
code, _, err = run(
|
|
40
|
+
{"tool_name": "exec_command", "tool_input": tool_input, "cwd": str(tmp_path)},
|
|
41
|
+
agent="codex",
|
|
42
|
+
)
|
|
43
|
+
assert code == 2 and "couldn't be read" in err
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
# P2 7: a non-string or blank Bash command isn't readable.
|
|
47
|
+
@pytest.mark.parametrize("command", [[], {"a": 1}, 5, "", " "])
|
|
48
|
+
def test_non_string_command_asks(tmp_path, command):
|
|
49
|
+
_, out, _ = run(
|
|
50
|
+
{"tool_name": "Bash", "tool_input": {"command": command}, "cwd": str(tmp_path)}
|
|
51
|
+
)
|
|
52
|
+
assert out["hookSpecificOutput"]["permissionDecision"] == "ask"
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
# P2 8: a write without a readable destination asks.
|
|
56
|
+
@pytest.mark.parametrize("tool", ["Write", "Edit", "MultiEdit"])
|
|
57
|
+
def test_write_without_destination_asks(tmp_path, tool):
|
|
58
|
+
_, out, _ = run(
|
|
59
|
+
{
|
|
60
|
+
"tool_name": tool,
|
|
61
|
+
"tool_input": {"content": "ordinary text"},
|
|
62
|
+
"cwd": str(tmp_path),
|
|
63
|
+
}
|
|
64
|
+
)
|
|
65
|
+
assert out["hookSpecificOutput"]["permissionDecision"] == "ask"
|
|
66
|
+
assert "destination" in out["hookSpecificOutput"]["permissionDecisionReason"]
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
# P2 3: example and backup files aren't the active configuration.
|
|
70
|
+
@pytest.mark.parametrize(
|
|
71
|
+
("command", "asks"),
|
|
72
|
+
[
|
|
73
|
+
("cat guard.toml.example", False),
|
|
74
|
+
("cp .claude/settings.json.backup /tmp/x", False),
|
|
75
|
+
("cat guard.toml", True),
|
|
76
|
+
("echo x >> .claude/settings.local.json", True),
|
|
77
|
+
],
|
|
78
|
+
)
|
|
79
|
+
def test_mention_needs_the_whole_config_name(tmp_path, command, asks):
|
|
80
|
+
from judgetap.guard.rules import check_command_writes
|
|
81
|
+
|
|
82
|
+
assert (check_command_writes(command, tmp_path) is not None) is asks
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
# P2 4: the repo config is parsed once per action.
|
|
86
|
+
def test_repo_config_parsed_once(tmp_path, monkeypatch):
|
|
87
|
+
import tomllib
|
|
88
|
+
|
|
89
|
+
(tmp_path / "guard.toml").write_text('[[rule]]\npattern = "x"\noutcome = "allow"\n')
|
|
90
|
+
calls = []
|
|
91
|
+
real = tomllib.load
|
|
92
|
+
monkeypatch.setattr(tomllib, "load", lambda fh: calls.append(1) or real(fh))
|
|
93
|
+
run(
|
|
94
|
+
{
|
|
95
|
+
"tool_name": "Bash",
|
|
96
|
+
"tool_input": {"command": "ls"},
|
|
97
|
+
"cwd": str(tmp_path),
|
|
98
|
+
"session_id": "s1",
|
|
99
|
+
}
|
|
100
|
+
)
|
|
101
|
+
assert len(calls) == 1 # ~/.judgetap/guard.toml doesn't exist here
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
# P2 6: the ignored-allow warning survives an engine error and a Codex deny.
|
|
105
|
+
def test_repo_warning_not_lost_behind_engine_error(tmp_path, monkeypatch):
|
|
106
|
+
monkeypatch.setenv("JUDGETAP_ENGINE", "nonsense")
|
|
107
|
+
(tmp_path / "guard.toml").write_text('[[rule]]\npattern = "x"\noutcome = "allow"\n')
|
|
108
|
+
_, out, _ = run(
|
|
109
|
+
{
|
|
110
|
+
"tool_name": "Bash",
|
|
111
|
+
"tool_input": {"command": "make x"},
|
|
112
|
+
"cwd": str(tmp_path),
|
|
113
|
+
"session_id": "s2",
|
|
114
|
+
}
|
|
115
|
+
)
|
|
116
|
+
msg = out["systemMessage"]
|
|
117
|
+
assert "ignored 1 allow rule" in msg and "nonsense" in msg
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def test_repo_warning_shown_on_codex_deny(tmp_path):
|
|
121
|
+
(tmp_path / "guard.toml").write_text('[[rule]]\npattern = "x"\noutcome = "allow"\n')
|
|
122
|
+
code, _, err = run(
|
|
123
|
+
{
|
|
124
|
+
"tool_name": "exec_command",
|
|
125
|
+
"tool_input": {"cmd": "git push --force"},
|
|
126
|
+
"cwd": str(tmp_path),
|
|
127
|
+
"session_id": "s3",
|
|
128
|
+
},
|
|
129
|
+
agent="codex",
|
|
130
|
+
)
|
|
131
|
+
assert code == 2 and "ignored 1 allow rule" in err
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
# P2 9: README describes the guard-config check.
|
|
135
|
+
def test_readme_mentions_guard_config_check():
|
|
136
|
+
from pathlib import Path
|
|
137
|
+
|
|
138
|
+
readme = (Path(__file__).parent.parent / "README.md").read_text()
|
|
139
|
+
assert "writes to the guard's own configuration" in readme
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|