judgetap 0.0.2.dev28__tar.gz → 0.0.2.dev30__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/.github/workflows/release.yml +25 -0
  2. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/PKG-INFO +1 -1
  3. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/docs/SPEC.md +1 -1
  4. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/pyproject.toml +1 -1
  5. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/release-please-config.json +1 -6
  6. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/__init__.py +1 -1
  7. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/guard/core.py +28 -6
  8. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/guard/hook.py +79 -14
  9. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/guard/rules.py +103 -1
  10. judgetap-0.0.2.dev30/tests/test_guard_trust.py +320 -0
  11. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/.github/workflows/ci.yml +0 -0
  12. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/.github/workflows/demo.yml +0 -0
  13. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/.gitignore +0 -0
  14. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/.python-version +0 -0
  15. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/.release-please-manifest.json +0 -0
  16. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/CONTRIBUTING.md +0 -0
  17. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/LICENSE +0 -0
  18. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/README.md +0 -0
  19. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/docs/demo.tape +0 -0
  20. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/_compat.py +0 -0
  21. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/api.py +0 -0
  22. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/cascade.py +0 -0
  23. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/cli.py +0 -0
  24. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/dashboard/__init__.py +0 -0
  25. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/dashboard/data.py +0 -0
  26. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/dashboard/page.html +0 -0
  27. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/dashboard/server.py +0 -0
  28. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/decision_log.py +0 -0
  29. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/engine.py +0 -0
  30. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/engines/__init__.py +0 -0
  31. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/engines/agentjev.py +0 -0
  32. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/engines/jev.py +0 -0
  33. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/engines/laya.py +0 -0
  34. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/engines/llm.py +0 -0
  35. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/errors.py +0 -0
  36. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/evaluate.py +0 -0
  37. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/guard/__init__.py +0 -0
  38. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/guard/install.py +0 -0
  39. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/guard/loop.py +0 -0
  40. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/guard/stop.py +0 -0
  41. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/py.typed +0 -0
  42. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/secrets.py +0 -0
  43. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/testing.py +0 -0
  44. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/src/judgetap/types.py +0 -0
  45. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/tests/test_api.py +0 -0
  46. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/tests/test_call_accounting.py +0 -0
  47. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/tests/test_calls.py +0 -0
  48. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/tests/test_cascade.py +0 -0
  49. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/tests/test_dashboard.py +0 -0
  50. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/tests/test_decision_log.py +0 -0
  51. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/tests/test_engine_jev.py +0 -0
  52. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/tests/test_engine_llm.py +0 -0
  53. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/tests/test_engine_local.py +0 -0
  54. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/tests/test_evaluate.py +0 -0
  55. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/tests/test_guard_agents.py +0 -0
  56. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/tests/test_guard_core.py +0 -0
  57. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/tests/test_guard_hook.py +0 -0
  58. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/tests/test_guard_loop.py +0 -0
  59. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/tests/test_guard_rules.py +0 -0
  60. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/tests/test_guard_stop.py +0 -0
  61. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/tests/test_questions.py +0 -0
  62. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/tests/test_robustness_68.py +0 -0
  63. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/tests/test_secrets.py +0 -0
  64. {judgetap-0.0.2.dev28 → judgetap-0.0.2.dev30}/uv.lock +0 -0
@@ -26,6 +26,7 @@ jobs:
26
26
  outputs:
27
27
  released: ${{ steps.rp.outputs.release_created }}
28
28
  tag: ${{ steps.rp.outputs.tag_name }}
29
+ pr_branch: ${{ steps.rp.outputs.pr && fromJSON(steps.rp.outputs.pr).headBranchName || '' }}
29
30
  steps:
30
31
  - id: rp
31
32
  uses: googleapis/release-please-action@v4
@@ -33,6 +34,30 @@ jobs:
33
34
  config-file: release-please-config.json
34
35
  manifest-file: .release-please-manifest.json
35
36
 
37
+ # release-please bumps pyproject.toml but can't edit uv.lock's package
38
+ # table, so relock the release PR's branch whenever release-please updates it.
39
+ relock:
40
+ needs: release-please
41
+ if: github.event_name == 'push' && needs.release-please.outputs.pr_branch != ''
42
+ runs-on: ubuntu-latest
43
+ permissions:
44
+ contents: write
45
+ steps:
46
+ - uses: actions/checkout@v4
47
+ with:
48
+ ref: ${{ needs.release-please.outputs.pr_branch }}
49
+ - uses: astral-sh/setup-uv@v6
50
+ - run: uv lock
51
+ - name: Commit the refreshed lockfile
52
+ env:
53
+ BRANCH: ${{ needs.release-please.outputs.pr_branch }}
54
+ run: |
55
+ git diff --quiet uv.lock && exit 0
56
+ git config user.name "github-actions[bot]"
57
+ git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
58
+ git commit -m "chore: refresh uv.lock for the release" uv.lock
59
+ git push origin "HEAD:refs/heads/$BRANCH"
60
+
36
61
  # Stable: the release-please release (created with GITHUB_TOKEN, which
37
62
  # doesn't trigger `release` events, so no double publish) or a release
38
63
  # published by hand in the GitHub UI.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: judgetap
3
- Version: 0.0.2.dev28
3
+ Version: 0.0.2.dev30
4
4
  Summary: Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development.
5
5
  Project-URL: Homepage, https://github.com/mergesafe-ai/judgetap
6
6
  Author-email: Omer Bar-Ness <omer@zsquared.io>
@@ -71,7 +71,7 @@ A pre-action hook for coding agents, built on the core.
71
71
 
72
72
  ## Decided: the guard's default engine (#8)
73
73
 
74
- No engine by default, and nothing downloaded or stored. Rules-only mode fails closed for shell commands (it asks when unsure); without an engine, file writes and edits are only checked for secrets and user rules. `judgetap guard install` uses an engine the user already has, in this order: `$JUDGETAP_ENGINE`, a `TYPESAFE_API_KEY` (Jev), a local AgentJev server on 127.0.0.1:8149. It records the choice as `engine = "..."` in `~/.judgetap/guard.toml`, because agents often run hooks without the user's shell environment. Install never stores a key by itself; when it detects `TYPESAFE_API_KEY` in the environment and runs in a terminal, it offers (opt-in) to copy it into the OS keychain, and engines read the environment first, then the keychain. A local model isn't the default because of the download (Laya pulls PyTorch; AgentJev needs its own server), and the user's LLM key isn't auto-picked because it adds seconds per guarded call. Either is one line in guard.toml.
74
+ No engine by default, and nothing downloaded or stored. Rules-only mode fails closed for shell commands (it asks when unsure); without an engine, file writes and edits are checked for secrets, user rules and writes to the guard's own configuration (guard.toml, agent hook settings), and any shell command that mentions one of those paths (or a word in it that resolves to one through a symlink) asks, since listing every way the shell can write a file isn't possible. `judgetap guard install` uses an engine the user already has, in this order: `$JUDGETAP_ENGINE`, a `TYPESAFE_API_KEY` (Jev), a local AgentJev server on 127.0.0.1:8149. It records the choice as `engine = "..."` in `~/.judgetap/guard.toml`, because agents often run hooks without the user's shell environment. Install never stores a key by itself; when it detects `TYPESAFE_API_KEY` in the environment and runs in a terminal, it offers (opt-in) to copy it into the OS keychain, and engines read the environment first, then the keychain. A local model isn't the default because of the download (Laya pulls PyTorch; AgentJev needs its own server), and the user's LLM key isn't auto-picked because it adds seconds per guarded call. Either is one line in guard.toml.
75
75
 
76
76
  **Experimental, opt-in: task-done check.** `judgetap guard install --with stop` adds a Claude Code `Stop` hook that asks the engine whether the user's task is actually finished. Only a confident "not done" (p(done) <= `stop_threshold`, default 0.15, set in `~/.judgetap/guard.toml`) blocks the stop, at most twice per session, and never while a previous block is being handled. It needs an engine and is off by default until an eval set shows it's reliable.
77
77
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "judgetap"
3
- version = "0.0.2.dev28"
3
+ version = "0.0.2.dev30"
4
4
  description = "Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development."
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -7,12 +7,7 @@
7
7
  "bump-patch-for-minor-pre-major": false,
8
8
  "changelog-path": "CHANGELOG.md",
9
9
  "extra-files": [
10
- "src/judgetap/__init__.py",
11
- {
12
- "type": "toml",
13
- "path": "uv.lock",
14
- "jsonpath": "$.package[?(@.name=='judgetap')].version"
15
- }
10
+ "src/judgetap/__init__.py"
16
11
  ]
17
12
  }
18
13
  },
@@ -23,7 +23,7 @@ from judgetap.errors import (
23
23
  )
24
24
  from judgetap.types import Decision, Question
25
25
 
26
- __version__ = "0.0.2.dev28" # x-release-please-version
26
+ __version__ = "0.0.2.dev30" # x-release-please-version
27
27
 
28
28
  __all__ = [
29
29
  "Cascade",
@@ -13,7 +13,9 @@ from judgetap.engine import Engine
13
13
  from judgetap.guard.rules import (
14
14
  Outcome,
15
15
  check_command,
16
+ check_command_writes,
16
17
  check_content,
18
+ check_path,
17
19
  rules_only_check,
18
20
  )
19
21
  from judgetap.types import Question
@@ -62,6 +64,7 @@ class Action:
62
64
  content: str | None = None
63
65
  task: str | None = None
64
66
  project_rules: str | None = None
67
+ unreadable: bool = False # a Bash call with no readable command
65
68
 
66
69
 
67
70
  @dataclass
@@ -85,8 +88,13 @@ class UserRule:
85
88
  reason: str
86
89
 
87
90
 
88
- def load_user_rules(path: Path) -> list[UserRule]:
89
- """[[rule]] tables from guard.toml: pattern (regex), outcome, reason."""
91
+ def load_user_rules(path: Path, *, trusted: bool = True) -> list[UserRule]:
92
+ """[[rule]] tables from guard.toml: pattern (regex), outcome, reason.
93
+
94
+ A repo's own guard.toml (trusted=False) may only tighten: its `allow`
95
+ rules are dropped, since anything in a cloned repo, or written by the
96
+ agent, could otherwise switch the built-in rules off. Only the user's
97
+ ~/.judgetap/guard.toml can allow. `dropped_allows(path)` reports them."""
90
98
  if not path.is_file():
91
99
  return []
92
100
  with path.open("rb") as fh:
@@ -98,6 +106,8 @@ def load_user_rules(path: Path) -> list[UserRule]:
98
106
  raise ValueError(
99
107
  f"{path}: outcome must be hold, ask or allow, not {outcome!r}"
100
108
  )
109
+ if outcome == "allow" and not trusted:
110
+ continue # a repo can't loosen the guard
101
111
  rules.append(
102
112
  UserRule(
103
113
  re.compile(entry["pattern"]),
@@ -117,7 +127,8 @@ def check(
117
127
  engine: Engine | None = None,
118
128
  user_rules: list[UserRule] | None = None,
119
129
  ) -> Verdict:
120
- # User rules first: an explicit "allow" is how a team overrides a built-in.
130
+ # User rules first: an explicit "allow" is how a team overrides a built-in,
131
+ # but only from the user's own config (see load_user_rules' `trusted`).
121
132
  subject = _subject(action)
122
133
  for rule in user_rules or []:
123
134
  if rule.pattern.search(subject):
@@ -126,9 +137,11 @@ def check(
126
137
  )
127
138
  hit = None
128
139
  if action.command is not None:
129
- hit = check_command(action.command, action.cwd)
130
- elif action.content is not None:
131
- hit = check_content(action.content)
140
+ hit = check_command(action.command, action.cwd) or check_command_writes(
141
+ action.command, action.cwd
142
+ )
143
+ else:
144
+ hit = check_content(action.content or "") or check_path(action.path, action.cwd)
132
145
  if hit:
133
146
  outcome, name, reason = hit
134
147
  return Verdict(outcome, "rules", reason, rule=name)
@@ -214,3 +227,12 @@ def project_rules(cwd: Path) -> str | None:
214
227
  if (directory / ".git").exists():
215
228
  break
216
229
  return None
230
+
231
+
232
+ def dropped_allows(path: Path) -> int:
233
+ """How many `allow` rules a repo guard.toml has (ignored when loaded untrusted)."""
234
+ if not path.is_file():
235
+ return 0
236
+ with path.open("rb") as fh:
237
+ data = tomllib.load(fh)
238
+ return sum(1 for e in data.get("rule", []) if e.get("outcome", "hold") == "allow")
@@ -9,6 +9,7 @@ from __future__ import annotations
9
9
 
10
10
  import json
11
11
  import os
12
+ import re
12
13
  import shlex
13
14
  import sys
14
15
  import time
@@ -19,7 +20,14 @@ from pathlib import Path
19
20
  from typing import Any
20
21
 
21
22
  from judgetap._compat import default_home, env
22
- from judgetap.guard.core import Action, Verdict, check, load_user_rules, project_rules
23
+ from judgetap.guard.core import (
24
+ Action,
25
+ Verdict,
26
+ check,
27
+ dropped_allows,
28
+ load_user_rules,
29
+ project_rules,
30
+ )
23
31
  from judgetap.guard.rules import redact
24
32
 
25
33
  TRANSCRIPT_SCAN_BYTES = 2 * 1024 * 1024
@@ -31,27 +39,47 @@ def home() -> Path:
31
39
  return Path(configured) if configured else default_home()
32
40
 
33
41
 
42
+ def _text(value: Any) -> str:
43
+ """Hook fields as text, whatever type arrived (None -> "")."""
44
+ return "" if value is None else value if isinstance(value, str) else str(value)
45
+
46
+
34
47
  def action_from_hook(payload: dict[str, Any]) -> Action | None:
48
+ """The action to check. Parsing never raises on odd field types; if the
49
+ extras (task, project rules) can't be read, the rules still run."""
35
50
  tool = payload.get("tool_name")
36
51
  if tool not in GUARDED_TOOLS:
37
52
  return None
38
- inp = payload.get("tool_input") or {}
39
- cwd = Path(payload.get("cwd") or os.getcwd())
53
+ inp = payload.get("tool_input")
54
+ inp = inp if isinstance(inp, dict) else {}
55
+ cwd = Path(_text(payload.get("cwd")) or os.getcwd())
40
56
  action = Action(tool=tool, cwd=cwd)
41
57
  if tool == "Bash":
42
- action.command = inp.get("command", "")
58
+ command = inp.get("command")
59
+ # None means the command couldn't be read: _decide asks rather than allows.
60
+ action.command = None if command is None else _text(command)
61
+ action.unreadable = command is None
43
62
  else:
44
- action.path = inp.get("file_path")
63
+ action.path = _text(inp.get("file_path")) or None
45
64
  if tool == "Write":
46
- action.content = inp.get("content", "")
65
+ action.content = _text(inp.get("content"))
47
66
  elif tool == "Edit":
48
- action.content = inp.get("new_string", "")
67
+ action.content = _text(inp.get("new_string"))
49
68
  else:
69
+ edits = inp.get("edits")
70
+ edits = edits if isinstance(edits, list) else []
50
71
  action.content = "\n".join(
51
- e.get("new_string", "") for e in inp.get("edits", [])
72
+ _text(e.get("new_string")) if isinstance(e, dict) else _text(e)
73
+ for e in edits
52
74
  )
53
- action.task = last_user_message(payload.get("transcript_path"))
54
- action.project_rules = project_rules(cwd)
75
+ try:
76
+ action.task = last_user_message(_text(payload.get("transcript_path")) or None)
77
+ except Exception: # noqa: BLE001 -- enrichment only: the rules still run
78
+ action.task = None
79
+ try:
80
+ action.project_rules = project_rules(cwd)
81
+ except Exception: # noqa: BLE001 -- enrichment only
82
+ action.project_rules = None
55
83
  return action
56
84
 
57
85
 
@@ -265,7 +293,7 @@ def run(
265
293
  if action is None:
266
294
  out = {"permission": "allow"} if agent == "cursor" else None
267
295
  else:
268
- verdict = _decide(action, start)
296
+ verdict = _decide(action, start, payload.get("session_id"))
269
297
  if record:
270
298
  try:
271
299
  log(action, verdict, payload.get("session_id"))
@@ -286,7 +314,35 @@ def run(
286
314
  return code
287
315
 
288
316
 
289
- def _decide(action: Action, start: float) -> Verdict:
317
+ def _warn_once(session: str | None, key: str) -> bool:
318
+ """True the first time `key` is warned about in this session."""
319
+ if not isinstance(session, str) or not re.fullmatch(
320
+ r"[A-Za-z0-9._-]{1,128}", session
321
+ ):
322
+ return True # no usable session id: warn rather than stay silent
323
+ marker = home() / "sessions" / f"{session}.warned"
324
+ try:
325
+ seen = set(marker.read_text().splitlines()) if marker.exists() else set()
326
+ if key in seen:
327
+ return False
328
+ marker.parent.mkdir(mode=0o700, parents=True, exist_ok=True)
329
+ fd = os.open(marker, os.O_WRONLY | os.O_APPEND | os.O_CREAT, 0o600)
330
+ with os.fdopen(fd, "a") as fh:
331
+ fh.write(key + "\n")
332
+ except OSError:
333
+ return True
334
+ return True
335
+
336
+
337
+ def _decide(action: Action, start: float, session: str | None = None) -> Verdict:
338
+ if getattr(action, "unreadable", False):
339
+ # A Bash call whose command can't be read: fail closed.
340
+ return Verdict(
341
+ "ask",
342
+ "rules",
343
+ "the command couldn't be read from the hook input",
344
+ rule="unreadable",
345
+ )
290
346
  try:
291
347
  engine = _engine()
292
348
  except Exception as err: # noqa: BLE001 -- bad engine config degrades to rules only
@@ -294,9 +350,18 @@ def _decide(action: Action, start: float) -> Verdict:
294
350
  else:
295
351
  engine_error = None
296
352
  rules, config_error = [], None
297
- for path in (home() / "guard.toml", action.cwd / "guard.toml"):
353
+ for path, trusted in (
354
+ (home() / "guard.toml", True),
355
+ (action.cwd / "guard.toml", False),
356
+ ):
298
357
  try:
299
- rules += load_user_rules(path)
358
+ rules += load_user_rules(path, trusted=trusted)
359
+ if (
360
+ not trusted
361
+ and (n := dropped_allows(path))
362
+ and _warn_once(session, f"repo-allow:{path}")
363
+ ):
364
+ config_error = f"ignored {n} allow rule(s) in {path}: a repo can only tighten the guard"
300
365
  except Exception as err: # noqa: BLE001 -- a bad config must not disable built-ins
301
366
  config_error = f"ignored {path}: {err}"
302
367
  verdict = check(action, engine, rules)
@@ -639,7 +639,15 @@ COMMAND_SECRETS = (
639
639
  re.compile(
640
640
  r"(?i)(?P<keep>(--)?(password|passwd|token|secret|api[-_]?key)[= ]\s*)[^\s'\"]+"
641
641
  ),
642
- re.compile(r"(?P<keep>\b[A-Z0-9_]*(TOKEN|SECRET|PASSWORD|API_KEY)=)\S+"),
642
+ # Env assignments with the keyword anywhere in the name:
643
+ # AWS_SECRET_ACCESS_KEY=, GH_TOKEN_RW=, PGPASSWORD=, DB_PRIVATE_KEY=.
644
+ # mysql/mariadb take the password attached: -phunter2 (a bare -p prompts).
645
+ re.compile(
646
+ r"(?P<keep>\b(mysql|mariadb|mysqldump|mysqladmin)\b[^;&|\n]*?\s-p)"
647
+ r"(?:'[^']*'|\"[^\"]*\"|[^\s'\"]+)"
648
+ ),
649
+ # sshpass -p <password>.
650
+ re.compile(r"(?P<keep>\bsshpass\s+-p\s*)(?:'[^']*'|\"[^\"]*\"|[^\s'\"]+)"),
643
651
  re.compile(r"(?P<keep>://[^:/\s@]+:)[^@\s]+(?=@)"),
644
652
  # Basic auth on the command line: curl -u user:pass, --user=user:pass.
645
653
  # Only user:pass values, so `sort -u file` and `git add -u` stay readable.
@@ -724,9 +732,30 @@ def _redact_segment(text: str) -> str:
724
732
  return "".join(out)
725
733
 
726
734
 
735
+ ENV_SECRET = re.compile(
736
+ r"(?i)\b[A-Z0-9_]*(TOKEN|SECRET|PASSWORD|PASSWD|API_KEY|ACCESS_KEY|PRIVATE_KEY)[A-Z0-9_]*="
737
+ )
738
+
739
+
740
+ def _redact_env_assignments(text: str) -> str:
741
+ """Mask the whole value of a secret-named assignment, quoted values and
742
+ escapes included (`DB_PRIVATE_KEY="two words"`), via the shell-word scanner."""
743
+ out, pos = [], 0
744
+ for m in ENV_SECRET.finditer(text):
745
+ if m.start() < pos:
746
+ continue
747
+ end = _shell_word_end(text, m.end())
748
+ if end > m.end():
749
+ out.append(text[pos : m.end()] + "[REDACTED]")
750
+ pos = end
751
+ out.append(text[pos:])
752
+ return "".join(out)
753
+
754
+
727
755
  def redact(text: str) -> str:
728
756
  """Mask credential-looking values so a command can be logged."""
729
757
  text = _redact_basic_auth(text)
758
+ text = _redact_env_assignments(text)
730
759
  for _name, pattern in SECRET_PATTERNS:
731
760
  text = pattern.sub("[REDACTED]", text)
732
761
  for pattern in COMMAND_SECRETS:
@@ -853,3 +882,76 @@ def rules_only_check(command: str) -> tuple[Outcome, str, str] | None:
853
882
  def _first_command_index(argv: list[str]) -> int:
854
883
  """Index of the command a wrapper chain finally runs."""
855
884
  return len(argv) - len(_unwrap(argv))
885
+
886
+
887
+ # Files that configure the guard (or the agent's hooks). An agent editing
888
+ # them could switch the guard off, so a person should see it first.
889
+ GUARD_CONFIG = re.compile(
890
+ r"(^|/)(guard\.toml|\.claude/settings[^/]*\.json|\.cursor/hooks\.json"
891
+ r"|\.codex/hooks\.json|\.codex/config\.toml)$"
892
+ )
893
+
894
+
895
+ def _is_guard_config(path: str, cwd: Path | None) -> bool:
896
+ """The literal path, and where it really points (symlinks followed,
897
+ relative paths taken from cwd), both checked."""
898
+ candidates = [path.replace("\\", "/")]
899
+ try:
900
+ p = Path(path).expanduser()
901
+ if cwd is not None and not p.is_absolute():
902
+ p = cwd / p
903
+ candidates.append(p.resolve(strict=False).as_posix())
904
+ except (OSError, RuntimeError, ValueError):
905
+ pass
906
+ return any(GUARD_CONFIG.search(c) for c in candidates)
907
+
908
+
909
+ def check_path(
910
+ path: str | None, cwd: Path | None = None
911
+ ) -> tuple[Outcome, str, str] | None:
912
+ """Ask before a write to the guard's own configuration."""
913
+ if path and _is_guard_config(path, cwd):
914
+ return "ask", "guard-config", f"edits the guard's own configuration ({path})"
915
+ return None
916
+
917
+
918
+ # Any mention of a guard-config file in a shell command asks. Parsing every
919
+ # way the shell can write a file (fd redirects, --target-directory, dd of=,
920
+ # python -c, ...) is an unbounded list; a mention is not. Reading the config
921
+ # (`cat guard.toml`) asks too: rare, and cheap to approve.
922
+ GUARD_CONFIG_MENTION = re.compile(
923
+ r"(?:guard\.toml|\.claude/settings[^\s'\"/]*\.json|\.cursor/hooks\.json"
924
+ r"|\.codex/hooks\.json|\.codex/config\.toml)"
925
+ )
926
+
927
+
928
+ REDIRECT_PREFIX = re.compile(r"^(?:\d*|&)(?:>>?|<>?)\|?")
929
+
930
+
931
+ def _words(command: str) -> list[str]:
932
+ """Every shell word, with redirections split off even when attached."""
933
+ spaced = re.sub(r"(\d*|&)(>>?|<>?)\|?", lambda m: " " + m.group(0) + " ", command)
934
+ words: list[str] = []
935
+ for argv in _commands(spaced, unwrap=False) or []:
936
+ words += argv
937
+ return words
938
+
939
+
940
+ def check_command_writes(
941
+ command: str, cwd: Path | None = None
942
+ ) -> tuple[Outcome, str, str] | None:
943
+ """Ask when a shell command mentions the guard's own configuration, or
944
+ a word in it resolves (symlinks followed) to one."""
945
+ if GUARD_CONFIG_MENTION.search(command.replace("\\", "/")):
946
+ return "ask", "guard-config", "touches the guard's own configuration"
947
+ # Every word, with redirection operators peeled off (`2>alias`, `&>>x`,
948
+ # `>|y`) and `--opt=` prefixes dropped, is resolved: a symlink with any
949
+ # name can point at the config.
950
+ for word in _words(command):
951
+ word = REDIRECT_PREFIX.sub("", word)
952
+ # Both the whole word (a file may be named `a=b`) and the value of an
953
+ # `--opt=value` / `of=value` word.
954
+ for target in {word, word.split("=", 1)[-1]}:
955
+ if target and _is_guard_config(target, cwd):
956
+ return "ask", "guard-config", "touches the guard's own configuration"
957
+ return None
@@ -0,0 +1,320 @@
1
+ """Audit fixes: repo guard.toml can't loosen (#64), odd input doesn't fail
2
+ open (#65), common secrets are redacted (#66)."""
3
+
4
+ import io
5
+ import json
6
+
7
+ import pytest
8
+
9
+ from judgetap.guard import hook
10
+ from judgetap.guard.rules import redact
11
+
12
+
13
+ @pytest.fixture(autouse=True)
14
+ def _home(tmp_path, monkeypatch):
15
+ monkeypatch.setenv("JUDGETAP_HOME", str(tmp_path / "home"))
16
+ monkeypatch.delenv("JUDGETAP_ENGINE", raising=False)
17
+ monkeypatch.delenv("SNAPJUDGE_ENGINE", raising=False)
18
+
19
+
20
+ def run(payload):
21
+ out = io.StringIO()
22
+ hook.run(io.StringIO(json.dumps(payload)), out)
23
+ return json.loads(out.getvalue()) if out.getvalue() else {}
24
+
25
+
26
+ def decision(out):
27
+ return out.get("hookSpecificOutput", {}).get("permissionDecision", "allow")
28
+
29
+
30
+ def bash(cmd, cwd):
31
+ return {"tool_name": "Bash", "tool_input": {"command": cmd}, "cwd": str(cwd)}
32
+
33
+
34
+ # --- #64 ------------------------------------------------------------------
35
+
36
+
37
+ def test_repo_guard_toml_cannot_allow_everything(tmp_path):
38
+ (tmp_path / "guard.toml").write_text(
39
+ '[[rule]]\npattern = ".*"\noutcome = "allow"\n'
40
+ )
41
+ out = run(bash("rm -rf ~", tmp_path))
42
+ assert decision(out) == "deny"
43
+ assert "repo can only tighten" in out.get("systemMessage", "")
44
+
45
+
46
+ def test_repo_guard_toml_can_still_tighten(tmp_path):
47
+ (tmp_path / "guard.toml").write_text(
48
+ '[[rule]]\npattern = "kubectl"\noutcome = "hold"\n'
49
+ )
50
+ assert decision(run(bash("kubectl get pods", tmp_path))) == "deny"
51
+
52
+
53
+ def test_user_guard_toml_can_allow(tmp_path):
54
+ home = tmp_path / "home"
55
+ home.mkdir()
56
+ (home / "guard.toml").write_text(
57
+ '[[rule]]\npattern = "git push origin main"\noutcome = "allow"\n'
58
+ )
59
+ assert decision(run(bash("git push origin main", tmp_path))) == "allow"
60
+
61
+
62
+ @pytest.mark.parametrize(
63
+ "path",
64
+ [
65
+ "guard.toml",
66
+ "/r/.claude/settings.json",
67
+ "/r/.claude/settings.local.json",
68
+ "/r/.cursor/hooks.json",
69
+ "/r/.codex/hooks.json",
70
+ "/r/.codex/config.toml",
71
+ ],
72
+ )
73
+ def test_writing_guard_config_asks(tmp_path, path):
74
+ out = run(
75
+ {
76
+ "tool_name": "Write",
77
+ "tool_input": {"file_path": path, "content": "x"},
78
+ "cwd": str(tmp_path),
79
+ }
80
+ )
81
+ assert decision(out) == "ask"
82
+
83
+
84
+ def test_ordinary_writes_are_not_affected(tmp_path):
85
+ out = run(
86
+ {
87
+ "tool_name": "Write",
88
+ "tool_input": {"file_path": "src/app.toml", "content": "x"},
89
+ "cwd": str(tmp_path),
90
+ }
91
+ )
92
+ assert decision(out) == "allow"
93
+
94
+
95
+ # --- #65 ------------------------------------------------------------------
96
+
97
+
98
+ def test_null_multiedit_value_still_runs_the_rules(tmp_path):
99
+ secret = "AKIA" + "B" * 16
100
+ payload = {
101
+ "tool_name": "MultiEdit",
102
+ "tool_input": {
103
+ "file_path": "/x",
104
+ "edits": [{"old_string": "a", "new_string": None}, {"new_string": secret}],
105
+ },
106
+ "cwd": str(tmp_path),
107
+ }
108
+ out = run(payload)
109
+ assert "failed and allowed" not in out.get("systemMessage", "")
110
+ assert decision(out) == "deny" # the secret in the second edit is still caught
111
+
112
+
113
+ def test_audit_repro_null_new_string_is_not_a_crash(tmp_path):
114
+ payload = {
115
+ "tool_name": "MultiEdit",
116
+ "tool_input": {
117
+ "file_path": "/x",
118
+ "edits": [{"old_string": "a", "new_string": None}],
119
+ },
120
+ "cwd": str(tmp_path),
121
+ }
122
+ assert "failed and allowed" not in run(payload).get("systemMessage", "")
123
+
124
+
125
+ def test_unreadable_bash_command_asks(tmp_path):
126
+ out = run({"tool_name": "Bash", "tool_input": {}, "cwd": str(tmp_path)})
127
+ assert decision(out) == "ask"
128
+
129
+
130
+ def test_non_string_fields_are_coerced(tmp_path):
131
+ out = run(
132
+ {
133
+ "tool_name": "Bash",
134
+ "tool_input": {"command": ["git", "push", "-f"]},
135
+ "cwd": str(tmp_path),
136
+ }
137
+ )
138
+ assert "failed and allowed" not in out.get("systemMessage", "")
139
+
140
+
141
+ def test_enrichment_failure_still_runs_rules(tmp_path, monkeypatch):
142
+ monkeypatch.setattr(hook, "last_user_message", lambda p: 1 / 0)
143
+ monkeypatch.setattr(hook, "project_rules", lambda p: 1 / 0)
144
+ assert decision(run(bash("git push --force", tmp_path))) == "deny"
145
+
146
+
147
+ # --- #66 ------------------------------------------------------------------
148
+
149
+
150
+ @pytest.mark.parametrize(
151
+ ("command", "leak"),
152
+ [
153
+ ("AWS_SECRET_ACCESS_KEY=abcd1234 aws s3 ls", "abcd1234"),
154
+ ("export AWS_SECRET_ACCESS_KEY=abcd1234", "abcd1234"),
155
+ ("DB_PRIVATE_KEY_PEM=zzz9 ./run", "zzz9"),
156
+ ("PGPASSWORD=s3cret psql -h db", "s3cret"),
157
+ ("mysql -u root -phunter2 db", "hunter2"),
158
+ ("mysqldump -h h -u u -pverysecret app", "verysecret"),
159
+ ("sshpass -p hunter2 ssh host", "hunter2"),
160
+ ("sshpass -phunter2 ssh host", "hunter2"),
161
+ ],
162
+ )
163
+ def test_common_secrets_are_redacted(command, leak):
164
+ assert leak not in redact(command)
165
+
166
+
167
+ @pytest.mark.parametrize(
168
+ "command", ["git add -p", "mkdir -p build/out", "cp -p a b", "mysql -p db"]
169
+ )
170
+ def test_harmless_dash_p_is_kept(command):
171
+ assert redact(command) == command
172
+
173
+
174
+ def test_symlinked_path_to_guard_config_asks(tmp_path):
175
+ from judgetap.guard.rules import check_path
176
+
177
+ (tmp_path / "guard.toml").write_text("")
178
+ (tmp_path / "guard-alias").symlink_to(tmp_path / "guard.toml")
179
+ assert check_path("guard-alias", tmp_path)[1] == "guard-config"
180
+ assert check_path(str(tmp_path / "guard-alias"))[1] == "guard-config"
181
+ assert check_path("notes.md", tmp_path) is None
182
+
183
+
184
+ @pytest.mark.parametrize(
185
+ ("command", "leak"),
186
+ [
187
+ ("db_private_key=zzz ./run", "zzz"),
188
+ ("aws_secret_access_key=abc123 aws s3 ls", "abc123"),
189
+ ("sshpass -p 'hunter2' ssh host", "hunter2"),
190
+ ('sshpass -p "pa ss" ssh host', "pa ss"),
191
+ ("mysql -u root -p'hunter2' db", "hunter2"),
192
+ ('mysql -u root -p"pw 1" db', "pw 1"),
193
+ ],
194
+ )
195
+ def test_round2_redaction(command, leak):
196
+ from judgetap.guard.rules import redact
197
+
198
+ assert leak not in redact(command)
199
+
200
+
201
+ @pytest.mark.parametrize(
202
+ "command",
203
+ [
204
+ "printf x > .claude/settings.json",
205
+ "echo '{}' >> guard.toml",
206
+ "cat new | tee .cursor/hooks.json",
207
+ "cp evil.toml guard.toml",
208
+ "mv x .codex/hooks.json",
209
+ "sed -i 's/ask/allow/' guard.toml",
210
+ ],
211
+ )
212
+ def test_bash_writes_to_guard_config_ask(tmp_path, command):
213
+ from judgetap.guard.core import Action, check
214
+
215
+ v = check(Action(tool="Bash", cwd=tmp_path, command=command), None)
216
+ assert (v.outcome, v.rule) == ("ask", "guard-config")
217
+
218
+
219
+ def test_ordinary_redirects_are_untouched(tmp_path):
220
+ from judgetap.guard.rules import check_command_writes
221
+
222
+ assert check_command_writes("echo hi > out.txt; cp a b", tmp_path) is None
223
+
224
+
225
+ def test_repo_allow_warning_once_per_session(tmp_path, monkeypatch):
226
+ import io
227
+ import json
228
+
229
+ from judgetap.guard import hook
230
+
231
+ monkeypatch.setenv("JUDGETAP_HOME", str(tmp_path / "home"))
232
+ monkeypatch.delenv("JUDGETAP_ENGINE", raising=False)
233
+ (tmp_path / "guard.toml").write_text('[[rule]]\npattern = "x"\noutcome = "allow"\n')
234
+
235
+ def run():
236
+ out = io.StringIO()
237
+ payload = {
238
+ "tool_name": "Bash",
239
+ "tool_input": {"command": "make test"},
240
+ "cwd": str(tmp_path),
241
+ "session_id": "s1",
242
+ }
243
+ hook.run(io.StringIO(json.dumps(payload)), out)
244
+ return json.loads(out.getvalue()) if out.getvalue() else {}
245
+
246
+ assert "ignored" in run().get("systemMessage", "")
247
+ assert "systemMessage" not in run()
248
+
249
+
250
+ @pytest.mark.parametrize(
251
+ "command",
252
+ [
253
+ "echo x 2>guard.toml",
254
+ "cp settings.json --target-directory=.claude && mv .claude/settings.json .claude/settings.local.json",
255
+ "dd if=x of=.codex/config.toml",
256
+ "python3 -c \"open('guard.toml','w').write('')\"",
257
+ "cat guard.toml",
258
+ ],
259
+ )
260
+ def test_any_mention_of_guard_config_asks(command, tmp_path):
261
+ from judgetap.guard.rules import check_command_writes
262
+
263
+ hit = check_command_writes(command, tmp_path)
264
+ assert hit is not None and hit[0] == "ask"
265
+
266
+
267
+ def test_symlinked_config_in_a_shell_word_asks(tmp_path):
268
+ from judgetap.guard.rules import check_command_writes
269
+
270
+ (tmp_path / "guard.toml").write_text("")
271
+ (tmp_path / "alias.cfg").symlink_to(tmp_path / "guard.toml")
272
+ assert check_command_writes("echo x > alias.cfg", tmp_path) is not None
273
+
274
+
275
+ def test_unrelated_commands_do_not_ask(tmp_path):
276
+ from judgetap.guard.rules import check_command_writes
277
+
278
+ assert check_command_writes("echo hi > notes.txt && cp a.py b.py", tmp_path) is None
279
+
280
+
281
+ @pytest.mark.parametrize(
282
+ ("command", "leak"),
283
+ [
284
+ ('DB_PRIVATE_KEY="two words" ./run', "words"),
285
+ ("api_token='a b' make", "a b"),
286
+ ('X_SECRET="q\\"x" y', 'x"'),
287
+ ],
288
+ )
289
+ def test_quoted_env_values_fully_redacted(command, leak):
290
+ from judgetap.guard.rules import redact
291
+
292
+ assert leak not in redact(command)
293
+
294
+
295
+ @pytest.mark.parametrize(
296
+ "command",
297
+ [
298
+ "echo x 2>alias.cfg",
299
+ "echo x 2>alias",
300
+ "echo x &>>alias",
301
+ "echo x >|alias",
302
+ "cat < alias",
303
+ ],
304
+ )
305
+ def test_redirects_to_symlinked_config_ask(command, tmp_path):
306
+ from judgetap.guard.rules import check_command_writes
307
+
308
+ (tmp_path / "guard.toml").write_text("")
309
+ for name in ("alias.cfg", "alias"):
310
+ (tmp_path / name).symlink_to(tmp_path / "guard.toml")
311
+ assert check_command_writes(command, tmp_path) is not None
312
+
313
+
314
+ def test_symlink_with_equals_in_its_name_asks(tmp_path):
315
+ from judgetap.guard.rules import check_command_writes
316
+
317
+ (tmp_path / "guard.toml").write_text("")
318
+ (tmp_path / "alias=cfg").symlink_to(tmp_path / "guard.toml")
319
+ assert check_command_writes("echo x >alias=cfg", tmp_path) is not None
320
+ assert check_command_writes("dd if=x of=alias=cfg", tmp_path) is not None
File without changes
File without changes
File without changes