agent-autoguard 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_autoguard-0.1.0/LICENSE +21 -0
- agent_autoguard-0.1.0/PKG-INFO +207 -0
- agent_autoguard-0.1.0/README.md +188 -0
- agent_autoguard-0.1.0/agent_autoguard.egg-info/PKG-INFO +207 -0
- agent_autoguard-0.1.0/agent_autoguard.egg-info/SOURCES.txt +24 -0
- agent_autoguard-0.1.0/agent_autoguard.egg-info/dependency_links.txt +1 -0
- agent_autoguard-0.1.0/agent_autoguard.egg-info/entry_points.txt +2 -0
- agent_autoguard-0.1.0/agent_autoguard.egg-info/top_level.txt +1 -0
- agent_autoguard-0.1.0/autoguard/__init__.py +6 -0
- agent_autoguard-0.1.0/autoguard/__main__.py +5 -0
- agent_autoguard-0.1.0/autoguard/cli.py +102 -0
- agent_autoguard-0.1.0/autoguard/console/__init__.py +0 -0
- agent_autoguard-0.1.0/autoguard/console/index.html +326 -0
- agent_autoguard-0.1.0/autoguard/console/server.py +88 -0
- agent_autoguard-0.1.0/autoguard/demo.py +49 -0
- agent_autoguard-0.1.0/autoguard/escalate.py +52 -0
- agent_autoguard-0.1.0/autoguard/evals.py +136 -0
- agent_autoguard-0.1.0/autoguard/guard.py +147 -0
- agent_autoguard-0.1.0/autoguard/hooks/__init__.py +0 -0
- agent_autoguard-0.1.0/autoguard/hooks/claude_code.py +88 -0
- agent_autoguard-0.1.0/autoguard/jev.py +106 -0
- agent_autoguard-0.1.0/autoguard/log.py +33 -0
- agent_autoguard-0.1.0/autoguard/policy.py +44 -0
- agent_autoguard-0.1.0/pyproject.toml +34 -0
- agent_autoguard-0.1.0/setup.cfg +4 -0
- agent_autoguard-0.1.0/tests/test_autoguard.py +160 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Roshan Chandna
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: agent-autoguard
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A real-time tool-call firewall for AI agents, built on TypeSafe's Jev.
|
|
5
|
+
Author: Roshan Chandna
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/rchandnaWUSTL/auto-guard
|
|
8
|
+
Project-URL: Issues, https://github.com/rchandnaWUSTL/auto-guard/issues
|
|
9
|
+
Keywords: ai-agents,claude-code,guardrails,tool-calls,jev,typesafe
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Topic :: Security
|
|
15
|
+
Requires-Python: >=3.9
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Dynamic: license-file
|
|
19
|
+
|
|
20
|
+
# Auto-Guard
|
|
21
|
+
|
|
22
|
+
**A safety check on every tool call your AI agent makes.**
|
|
23
|
+
|
|
24
|
+
Auto-Guard runs before an agent's tool call executes. It sends the proposed call and the user's request to [Jev](https://typesafe.ai), TypeSafe's fast decision model, and gets back one of three verdicts:
|
|
25
|
+
|
|
26
|
+
| Verdict | What happens |
|
|
27
|
+
|---|---|
|
|
28
|
+
| 🟢 **allow** | The call runs normally. |
|
|
29
|
+
| 🟡 **escalate** | You're asked to approve it, with a one-line reason written by an LLM. |
|
|
30
|
+
| 🔴 **block** | The call never runs. The agent is told why and picks another approach. |
|
|
31
|
+
|
|
32
|
+
A check takes about 0.4s and costs about $0.000025.
|
|
33
|
+
|
|
34
|
+
▶️ **Demo:** [what it does](https://github.com/rchandnaWUSTL/auto-guard/blob/main/demo/out/auto-guard.mp4) · [how it works](https://github.com/rchandnaWUSTL/auto-guard/blob/main/demo/out/auto-guard-under-the-hood.mp4)
|
|
35
|
+
|
|
36
|
+
---
|
|
37
|
+
|
|
38
|
+
## Quick start (Claude Code)
|
|
39
|
+
|
|
40
|
+
**1. Install**
|
|
41
|
+
|
|
42
|
+
```bash
|
|
43
|
+
pip install agent-autoguard
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
This needs Python 3.9+. There are no other dependencies.
|
|
47
|
+
|
|
48
|
+
**2. Add an API key.** An [OpenRouter key](https://openrouter.ai/keys) is the easiest option:
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
mkdir -p ~/.autoguard && echo 'OPENROUTER_API_KEY=sk-or-...' >> ~/.autoguard/env
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
**3. Turn it on** in your project. Add `--user` to turn it on for every project:
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
autoguard install
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
That's it. Claude Code now checks every `Bash`, `Edit`, `Write`, `MultiEdit`, `NotebookEdit` and `WebFetch` call before it runs.
|
|
61
|
+
|
|
62
|
+
To confirm it works:
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
autoguard check
|
|
66
|
+
# BLOCK 402ms $0.000025 rm -rf /tmp/build && rm -rf ~/.aws
|
|
67
|
+
# ALLOW 402ms $0.000025 ls -la build/
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
To watch decisions live, run `autoguard console` and open http://127.0.0.1:8787.
|
|
71
|
+
|
|
72
|
+
> Auto-Guard only adds restrictions. When it allows a call, Claude Code's normal permission rules still apply, and it never auto-approves anything.
|
|
73
|
+
|
|
74
|
+
---
|
|
75
|
+
|
|
76
|
+
## Use it in any agent
|
|
77
|
+
|
|
78
|
+
```python
|
|
79
|
+
from autoguard import guard
|
|
80
|
+
|
|
81
|
+
decision = guard("bash", "rm -rf ./build && rm -rf ~/.aws", task="clean up the build directory")
|
|
82
|
+
|
|
83
|
+
decision.action # "block"
|
|
84
|
+
decision.reasons # ["risk 3.8/4 is critical", "destructive (0.95) and outside the task (in-scope 0.18)"]
|
|
85
|
+
decision.rationale # LLM-written reason (escalations only)
|
|
86
|
+
decision.latency_ms # 419.1
|
|
87
|
+
decision.cost # 0.0000252
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
Call `guard(tool, args, task="", plan="")` after your agent picks a tool and before it runs the tool. It works with any framework. `args` can be a string or a dict.
|
|
91
|
+
|
|
92
|
+
---
|
|
93
|
+
|
|
94
|
+
## How it decides
|
|
95
|
+
|
|
96
|
+
Each check is **one Jev request that asks five questions at once**:
|
|
97
|
+
|
|
98
|
+
| Question | Answer type | Asks |
|
|
99
|
+
|---|---|---|
|
|
100
|
+
| `destructive` | probability | Does it irreversibly delete, overwrite or leak data? |
|
|
101
|
+
| `in_scope` | probability | Is it what the user asked for? |
|
|
102
|
+
| `sensitive` | probability | Does it touch credentials, secrets, production or payments? |
|
|
103
|
+
| `action_class` | choice | read · local-write · network · destructive · privileged |
|
|
104
|
+
| `risk` | 0–4 scale | none · low · medium · high · critical |
|
|
105
|
+
|
|
106
|
+
A small policy turns the answers into a verdict:
|
|
107
|
+
|
|
108
|
+
- **Block** when risk is critical, or when the call is destructive, out of scope *and* high-risk.
|
|
109
|
+
- **Escalate** when risk is high, the call touches sensitive systems, the action is privileged, or Jev isn't confident.
|
|
110
|
+
- **Allow** everything else.
|
|
111
|
+
|
|
112
|
+
If Jev can't be reached, the call is escalated. You can change that.
|
|
113
|
+
|
|
114
|
+
---
|
|
115
|
+
|
|
116
|
+
## Configuration
|
|
117
|
+
|
|
118
|
+
**API keys.** Auto-Guard checks environment variables first. It then reads `.env.local` or `.env` files (in the current folder or any parent) and `~/.autoguard/env`.
|
|
119
|
+
|
|
120
|
+
| Variable | Get one | Used for |
|
|
121
|
+
|---|---|---|
|
|
122
|
+
| `OPENROUTER_API_KEY` | [openrouter.ai/keys](https://openrouter.ai/keys) | Jev checks + written escalation reasons |
|
|
123
|
+
| `TYPESAFE_API_KEY` | [console.typesafe.ai](https://console.typesafe.ai/keys) (early access) | Jev checks, direct from TypeSafe |
|
|
124
|
+
|
|
125
|
+
If both keys are set, OpenRouter is used. With only a TypeSafe key, escalations still work but come without a written reason.
|
|
126
|
+
|
|
127
|
+
**Policy.** To override any default, put a JSON file at `.autoguard.json` (per project) or `~/.autoguard/policy.json` (global):
|
|
128
|
+
|
|
129
|
+
```json
|
|
130
|
+
{
|
|
131
|
+
"block_risk": 3.5,
|
|
132
|
+
"escalate_risk": 2.5,
|
|
133
|
+
"escalate_sensitive": 0.7,
|
|
134
|
+
"on_error": "escalate",
|
|
135
|
+
"escalation_model": "anthropic/claude-haiku-4.5",
|
|
136
|
+
"spend_cap_usd": 2.5
|
|
137
|
+
}
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
All settings and their defaults are in [`autoguard/policy.py`](https://github.com/rchandnaWUSTL/auto-guard/blob/main/autoguard/policy.py).
|
|
141
|
+
|
|
142
|
+
**Log.** Every decision is appended to `~/.autoguard/decisions.jsonl`. To log somewhere else, set `AUTOGUARD_LOG`. Once the logged spend reaches `spend_cap_usd`, escalations stop calling the LLM.
|
|
143
|
+
|
|
144
|
+
---
|
|
145
|
+
|
|
146
|
+
## Commands
|
|
147
|
+
|
|
148
|
+
| Command | Does |
|
|
149
|
+
|---|---|
|
|
150
|
+
| `autoguard install [--user]` | Adds the Claude Code hook. Your existing settings are kept, and running it twice is safe. |
|
|
151
|
+
| `autoguard check` | Runs two sample calls to verify your key |
|
|
152
|
+
| `autoguard console` | Opens the live decision dashboard |
|
|
153
|
+
| `autoguard demo [block\|escalate\|all]` | Runs the scripted demo calls through the real guard |
|
|
154
|
+
| `autoguard eval` | Scores the policy on the labeled test calls (run it from a clone of this repo) |
|
|
155
|
+
|
|
156
|
+
---
|
|
157
|
+
|
|
158
|
+
## Results
|
|
159
|
+
|
|
160
|
+
Held-out test set of 20 calls, never used for tuning ([`evals/holdout.jsonl`](https://github.com/rchandnaWUSTL/auto-guard/blob/main/evals/holdout.jsonl)):
|
|
161
|
+
|
|
162
|
+
| | Safe allowed | Dangerous caught | Ambiguous escalated | Median latency | Cost per check |
|
|
163
|
+
|---|---|---|---|---|---|
|
|
164
|
+
| **Auto-Guard (Jev)** | **8/8** | **8/8** | **4/4** | **428ms** | **$0.000025** |
|
|
165
|
+
| Claude Haiku 4.5 as the guard | 8/8 | 8/8 | 3/4 | 823ms | $0.00012 |
|
|
166
|
+
| Claude Sonnet 5 as the guard | 7/8 | 5/8 | 1/4 | 1,992ms | $0.00032 |
|
|
167
|
+
|
|
168
|
+
On the 61-call tuning set, it allowed 26/26 safe calls and caught 25/25 dangerous ones. The thresholds were tuned on that set.
|
|
169
|
+
|
|
170
|
+
Latency was measured from a laptop via OpenRouter, with a new connection per call (the same as the Claude Code hook). Calling `guard()` repeatedly in one process reuses the connection, which brings the median to about 310–360ms.
|
|
171
|
+
|
|
172
|
+
---
|
|
173
|
+
|
|
174
|
+
## Limits
|
|
175
|
+
|
|
176
|
+
- **It's a filter, not a sandbox.** It will sometimes miss things. Keep backups, scoped credentials and sandboxing.
|
|
177
|
+
- **"In scope" is its weakest signal.** Jev sometimes misjudges multi-part requests. So scope only counts when the risk is also high.
|
|
178
|
+
- **It can block things you asked for.** "Delete my old AWS config" gets blocked because deleting credentials is high-risk. Run those commands yourself.
|
|
179
|
+
- **Each check adds about 0.3–0.6s.** That goes unnoticed next to an agent's own model calls, but it isn't free.
|
|
180
|
+
|
|
181
|
+
---
|
|
182
|
+
|
|
183
|
+
## Development
|
|
184
|
+
|
|
185
|
+
```bash
|
|
186
|
+
python3 -m unittest discover tests # offline tests, no API calls
|
|
187
|
+
python3 -m autoguard eval # re-score using cached Jev answers (free)
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
The demo videos are generated from recorded, real Jev responses. The same inputs always produce identical files:
|
|
191
|
+
|
|
192
|
+
```bash
|
|
193
|
+
cd demo && npm install && npm run video # writes demo/out/*.mp4
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
```
|
|
197
|
+
autoguard/ guard(), policy, Jev client, Claude Code hook, CLI, live console
|
|
198
|
+
evals/ labeled test calls and cached Jev answers
|
|
199
|
+
tests/ unit tests
|
|
200
|
+
demo/ demo scenes, video capture (Playwright) and editing (Remotion)
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
Built on [TypeSafe's Jev](https://typesafe.ai).
|
|
204
|
+
|
|
205
|
+
## License
|
|
206
|
+
|
|
207
|
+
MIT. See [LICENSE](https://github.com/rchandnaWUSTL/auto-guard/blob/main/LICENSE).
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
# Auto-Guard
|
|
2
|
+
|
|
3
|
+
**A safety check on every tool call your AI agent makes.**
|
|
4
|
+
|
|
5
|
+
Auto-Guard runs before an agent's tool call executes. It sends the proposed call and the user's request to [Jev](https://typesafe.ai), TypeSafe's fast decision model, and gets back one of three verdicts:
|
|
6
|
+
|
|
7
|
+
| Verdict | What happens |
|
|
8
|
+
|---|---|
|
|
9
|
+
| 🟢 **allow** | The call runs normally. |
|
|
10
|
+
| 🟡 **escalate** | You're asked to approve it, with a one-line reason written by an LLM. |
|
|
11
|
+
| 🔴 **block** | The call never runs. The agent is told why and picks another approach. |
|
|
12
|
+
|
|
13
|
+
A check takes about 0.4s and costs about $0.000025.
|
|
14
|
+
|
|
15
|
+
▶️ **Demo:** [what it does](https://github.com/rchandnaWUSTL/auto-guard/blob/main/demo/out/auto-guard.mp4) · [how it works](https://github.com/rchandnaWUSTL/auto-guard/blob/main/demo/out/auto-guard-under-the-hood.mp4)
|
|
16
|
+
|
|
17
|
+
---
|
|
18
|
+
|
|
19
|
+
## Quick start (Claude Code)
|
|
20
|
+
|
|
21
|
+
**1. Install**
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
pip install agent-autoguard
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
This needs Python 3.9+. There are no other dependencies.
|
|
28
|
+
|
|
29
|
+
**2. Add an API key.** An [OpenRouter key](https://openrouter.ai/keys) is the easiest option:
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
mkdir -p ~/.autoguard && echo 'OPENROUTER_API_KEY=sk-or-...' >> ~/.autoguard/env
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
**3. Turn it on** in your project. Add `--user` to turn it on for every project:
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
autoguard install
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
That's it. Claude Code now checks every `Bash`, `Edit`, `Write`, `MultiEdit`, `NotebookEdit` and `WebFetch` call before it runs.
|
|
42
|
+
|
|
43
|
+
To confirm it works:
|
|
44
|
+
|
|
45
|
+
```bash
|
|
46
|
+
autoguard check
|
|
47
|
+
# BLOCK 402ms $0.000025 rm -rf /tmp/build && rm -rf ~/.aws
|
|
48
|
+
# ALLOW 402ms $0.000025 ls -la build/
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
To watch decisions live, run `autoguard console` and open http://127.0.0.1:8787.
|
|
52
|
+
|
|
53
|
+
> Auto-Guard only adds restrictions. When it allows a call, Claude Code's normal permission rules still apply, and it never auto-approves anything.
|
|
54
|
+
|
|
55
|
+
---
|
|
56
|
+
|
|
57
|
+
## Use it in any agent
|
|
58
|
+
|
|
59
|
+
```python
|
|
60
|
+
from autoguard import guard
|
|
61
|
+
|
|
62
|
+
decision = guard("bash", "rm -rf ./build && rm -rf ~/.aws", task="clean up the build directory")
|
|
63
|
+
|
|
64
|
+
decision.action # "block"
|
|
65
|
+
decision.reasons # ["risk 3.8/4 is critical", "destructive (0.95) and outside the task (in-scope 0.18)"]
|
|
66
|
+
decision.rationale # LLM-written reason (escalations only)
|
|
67
|
+
decision.latency_ms # 419.1
|
|
68
|
+
decision.cost # 0.0000252
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
Call `guard(tool, args, task="", plan="")` after your agent picks a tool and before it runs the tool. It works with any framework. `args` can be a string or a dict.
|
|
72
|
+
|
|
73
|
+
---
|
|
74
|
+
|
|
75
|
+
## How it decides
|
|
76
|
+
|
|
77
|
+
Each check is **one Jev request that asks five questions at once**:
|
|
78
|
+
|
|
79
|
+
| Question | Answer type | Asks |
|
|
80
|
+
|---|---|---|
|
|
81
|
+
| `destructive` | probability | Does it irreversibly delete, overwrite or leak data? |
|
|
82
|
+
| `in_scope` | probability | Is it what the user asked for? |
|
|
83
|
+
| `sensitive` | probability | Does it touch credentials, secrets, production or payments? |
|
|
84
|
+
| `action_class` | choice | read · local-write · network · destructive · privileged |
|
|
85
|
+
| `risk` | 0–4 scale | none · low · medium · high · critical |
|
|
86
|
+
|
|
87
|
+
A small policy turns the answers into a verdict:
|
|
88
|
+
|
|
89
|
+
- **Block** when risk is critical, or when the call is destructive, out of scope *and* high-risk.
|
|
90
|
+
- **Escalate** when risk is high, the call touches sensitive systems, the action is privileged, or Jev isn't confident.
|
|
91
|
+
- **Allow** everything else.
|
|
92
|
+
|
|
93
|
+
If Jev can't be reached, the call is escalated. You can change that.
|
|
94
|
+
|
|
95
|
+
---
|
|
96
|
+
|
|
97
|
+
## Configuration
|
|
98
|
+
|
|
99
|
+
**API keys.** Auto-Guard checks environment variables first. It then reads `.env.local` or `.env` files (in the current folder or any parent) and `~/.autoguard/env`.
|
|
100
|
+
|
|
101
|
+
| Variable | Get one | Used for |
|
|
102
|
+
|---|---|---|
|
|
103
|
+
| `OPENROUTER_API_KEY` | [openrouter.ai/keys](https://openrouter.ai/keys) | Jev checks + written escalation reasons |
|
|
104
|
+
| `TYPESAFE_API_KEY` | [console.typesafe.ai](https://console.typesafe.ai/keys) (early access) | Jev checks, direct from TypeSafe |
|
|
105
|
+
|
|
106
|
+
If both keys are set, OpenRouter is used. With only a TypeSafe key, escalations still work but come without a written reason.
|
|
107
|
+
|
|
108
|
+
**Policy.** To override any default, put a JSON file at `.autoguard.json` (per project) or `~/.autoguard/policy.json` (global):
|
|
109
|
+
|
|
110
|
+
```json
|
|
111
|
+
{
|
|
112
|
+
"block_risk": 3.5,
|
|
113
|
+
"escalate_risk": 2.5,
|
|
114
|
+
"escalate_sensitive": 0.7,
|
|
115
|
+
"on_error": "escalate",
|
|
116
|
+
"escalation_model": "anthropic/claude-haiku-4.5",
|
|
117
|
+
"spend_cap_usd": 2.5
|
|
118
|
+
}
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
All settings and their defaults are in [`autoguard/policy.py`](https://github.com/rchandnaWUSTL/auto-guard/blob/main/autoguard/policy.py).
|
|
122
|
+
|
|
123
|
+
**Log.** Every decision is appended to `~/.autoguard/decisions.jsonl`. To log somewhere else, set `AUTOGUARD_LOG`. Once the logged spend reaches `spend_cap_usd`, escalations stop calling the LLM.
|
|
124
|
+
|
|
125
|
+
---
|
|
126
|
+
|
|
127
|
+
## Commands
|
|
128
|
+
|
|
129
|
+
| Command | Does |
|
|
130
|
+
|---|---|
|
|
131
|
+
| `autoguard install [--user]` | Adds the Claude Code hook. Your existing settings are kept, and running it twice is safe. |
|
|
132
|
+
| `autoguard check` | Runs two sample calls to verify your key |
|
|
133
|
+
| `autoguard console` | Opens the live decision dashboard |
|
|
134
|
+
| `autoguard demo [block\|escalate\|all]` | Runs the scripted demo calls through the real guard |
|
|
135
|
+
| `autoguard eval` | Scores the policy on the labeled test calls (run it from a clone of this repo) |
|
|
136
|
+
|
|
137
|
+
---
|
|
138
|
+
|
|
139
|
+
## Results
|
|
140
|
+
|
|
141
|
+
Held-out test set of 20 calls, never used for tuning ([`evals/holdout.jsonl`](https://github.com/rchandnaWUSTL/auto-guard/blob/main/evals/holdout.jsonl)):
|
|
142
|
+
|
|
143
|
+
| | Safe allowed | Dangerous caught | Ambiguous escalated | Median latency | Cost per check |
|
|
144
|
+
|---|---|---|---|---|---|
|
|
145
|
+
| **Auto-Guard (Jev)** | **8/8** | **8/8** | **4/4** | **428ms** | **$0.000025** |
|
|
146
|
+
| Claude Haiku 4.5 as the guard | 8/8 | 8/8 | 3/4 | 823ms | $0.00012 |
|
|
147
|
+
| Claude Sonnet 5 as the guard | 7/8 | 5/8 | 1/4 | 1,992ms | $0.00032 |
|
|
148
|
+
|
|
149
|
+
On the 61-call tuning set, it allowed 26/26 safe calls and caught 25/25 dangerous ones. The thresholds were tuned on that set.
|
|
150
|
+
|
|
151
|
+
Latency was measured from a laptop via OpenRouter, with a new connection per call (the same as the Claude Code hook). Calling `guard()` repeatedly in one process reuses the connection, which brings the median to about 310–360ms.
|
|
152
|
+
|
|
153
|
+
---
|
|
154
|
+
|
|
155
|
+
## Limits
|
|
156
|
+
|
|
157
|
+
- **It's a filter, not a sandbox.** It will sometimes miss things. Keep backups, scoped credentials and sandboxing.
|
|
158
|
+
- **"In scope" is its weakest signal.** Jev sometimes misjudges multi-part requests. So scope only counts when the risk is also high.
|
|
159
|
+
- **It can block things you asked for.** "Delete my old AWS config" gets blocked because deleting credentials is high-risk. Run those commands yourself.
|
|
160
|
+
- **Each check adds about 0.3–0.6s.** That goes unnoticed next to an agent's own model calls, but it isn't free.
|
|
161
|
+
|
|
162
|
+
---
|
|
163
|
+
|
|
164
|
+
## Development
|
|
165
|
+
|
|
166
|
+
```bash
|
|
167
|
+
python3 -m unittest discover tests # offline tests, no API calls
|
|
168
|
+
python3 -m autoguard eval # re-score using cached Jev answers (free)
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
The demo videos are generated from recorded, real Jev responses. The same inputs always produce identical files:
|
|
172
|
+
|
|
173
|
+
```bash
|
|
174
|
+
cd demo && npm install && npm run video # writes demo/out/*.mp4
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
```
|
|
178
|
+
autoguard/ guard(), policy, Jev client, Claude Code hook, CLI, live console
|
|
179
|
+
evals/ labeled test calls and cached Jev answers
|
|
180
|
+
tests/ unit tests
|
|
181
|
+
demo/ demo scenes, video capture (Playwright) and editing (Remotion)
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
Built on [TypeSafe's Jev](https://typesafe.ai).
|
|
185
|
+
|
|
186
|
+
## License
|
|
187
|
+
|
|
188
|
+
MIT. See [LICENSE](https://github.com/rchandnaWUSTL/auto-guard/blob/main/LICENSE).
|
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: agent-autoguard
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A real-time tool-call firewall for AI agents, built on TypeSafe's Jev.
|
|
5
|
+
Author: Roshan Chandna
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/rchandnaWUSTL/auto-guard
|
|
8
|
+
Project-URL: Issues, https://github.com/rchandnaWUSTL/auto-guard/issues
|
|
9
|
+
Keywords: ai-agents,claude-code,guardrails,tool-calls,jev,typesafe
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Topic :: Security
|
|
15
|
+
Requires-Python: >=3.9
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Dynamic: license-file
|
|
19
|
+
|
|
20
|
+
# Auto-Guard
|
|
21
|
+
|
|
22
|
+
**A safety check on every tool call your AI agent makes.**
|
|
23
|
+
|
|
24
|
+
Auto-Guard runs before an agent's tool call executes. It sends the proposed call and the user's request to [Jev](https://typesafe.ai), TypeSafe's fast decision model, and gets back one of three verdicts:
|
|
25
|
+
|
|
26
|
+
| Verdict | What happens |
|
|
27
|
+
|---|---|
|
|
28
|
+
| 🟢 **allow** | The call runs normally. |
|
|
29
|
+
| 🟡 **escalate** | You're asked to approve it, with a one-line reason written by an LLM. |
|
|
30
|
+
| 🔴 **block** | The call never runs. The agent is told why and picks another approach. |
|
|
31
|
+
|
|
32
|
+
A check takes about 0.4s and costs about $0.000025.
|
|
33
|
+
|
|
34
|
+
▶️ **Demo:** [what it does](https://github.com/rchandnaWUSTL/auto-guard/blob/main/demo/out/auto-guard.mp4) · [how it works](https://github.com/rchandnaWUSTL/auto-guard/blob/main/demo/out/auto-guard-under-the-hood.mp4)
|
|
35
|
+
|
|
36
|
+
---
|
|
37
|
+
|
|
38
|
+
## Quick start (Claude Code)
|
|
39
|
+
|
|
40
|
+
**1. Install**
|
|
41
|
+
|
|
42
|
+
```bash
|
|
43
|
+
pip install agent-autoguard
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
This needs Python 3.9+. There are no other dependencies.
|
|
47
|
+
|
|
48
|
+
**2. Add an API key.** An [OpenRouter key](https://openrouter.ai/keys) is the easiest option:
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
mkdir -p ~/.autoguard && echo 'OPENROUTER_API_KEY=sk-or-...' >> ~/.autoguard/env
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
**3. Turn it on** in your project. Add `--user` to turn it on for every project:
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
autoguard install
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
That's it. Claude Code now checks every `Bash`, `Edit`, `Write`, `MultiEdit`, `NotebookEdit` and `WebFetch` call before it runs.
|
|
61
|
+
|
|
62
|
+
To confirm it works:
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
autoguard check
|
|
66
|
+
# BLOCK 402ms $0.000025 rm -rf /tmp/build && rm -rf ~/.aws
|
|
67
|
+
# ALLOW 402ms $0.000025 ls -la build/
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
To watch decisions live, run `autoguard console` and open http://127.0.0.1:8787.
|
|
71
|
+
|
|
72
|
+
> Auto-Guard only adds restrictions. When it allows a call, Claude Code's normal permission rules still apply, and it never auto-approves anything.
|
|
73
|
+
|
|
74
|
+
---
|
|
75
|
+
|
|
76
|
+
## Use it in any agent
|
|
77
|
+
|
|
78
|
+
```python
|
|
79
|
+
from autoguard import guard
|
|
80
|
+
|
|
81
|
+
decision = guard("bash", "rm -rf ./build && rm -rf ~/.aws", task="clean up the build directory")
|
|
82
|
+
|
|
83
|
+
decision.action # "block"
|
|
84
|
+
decision.reasons # ["risk 3.8/4 is critical", "destructive (0.95) and outside the task (in-scope 0.18)"]
|
|
85
|
+
decision.rationale # LLM-written reason (escalations only)
|
|
86
|
+
decision.latency_ms # 419.1
|
|
87
|
+
decision.cost # 0.0000252
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
Call `guard(tool, args, task="", plan="")` after your agent picks a tool and before it runs the tool. It works with any framework. `args` can be a string or a dict.
|
|
91
|
+
|
|
92
|
+
---
|
|
93
|
+
|
|
94
|
+
## How it decides
|
|
95
|
+
|
|
96
|
+
Each check is **one Jev request that asks five questions at once**:
|
|
97
|
+
|
|
98
|
+
| Question | Answer type | Asks |
|
|
99
|
+
|---|---|---|
|
|
100
|
+
| `destructive` | probability | Does it irreversibly delete, overwrite or leak data? |
|
|
101
|
+
| `in_scope` | probability | Is it what the user asked for? |
|
|
102
|
+
| `sensitive` | probability | Does it touch credentials, secrets, production or payments? |
|
|
103
|
+
| `action_class` | choice | read · local-write · network · destructive · privileged |
|
|
104
|
+
| `risk` | 0–4 scale | none · low · medium · high · critical |
|
|
105
|
+
|
|
106
|
+
A small policy turns the answers into a verdict:
|
|
107
|
+
|
|
108
|
+
- **Block** when risk is critical, or when the call is destructive, out of scope *and* high-risk.
|
|
109
|
+
- **Escalate** when risk is high, the call touches sensitive systems, the action is privileged, or Jev isn't confident.
|
|
110
|
+
- **Allow** everything else.
|
|
111
|
+
|
|
112
|
+
If Jev can't be reached, the call is escalated. You can change that.
|
|
113
|
+
|
|
114
|
+
---
|
|
115
|
+
|
|
116
|
+
## Configuration
|
|
117
|
+
|
|
118
|
+
**API keys.** Auto-Guard checks environment variables first. It then reads `.env.local` or `.env` files (in the current folder or any parent) and `~/.autoguard/env`.
|
|
119
|
+
|
|
120
|
+
| Variable | Get one | Used for |
|
|
121
|
+
|---|---|---|
|
|
122
|
+
| `OPENROUTER_API_KEY` | [openrouter.ai/keys](https://openrouter.ai/keys) | Jev checks + written escalation reasons |
|
|
123
|
+
| `TYPESAFE_API_KEY` | [console.typesafe.ai](https://console.typesafe.ai/keys) (early access) | Jev checks, direct from TypeSafe |
|
|
124
|
+
|
|
125
|
+
If both keys are set, OpenRouter is used. With only a TypeSafe key, escalations still work but come without a written reason.
|
|
126
|
+
|
|
127
|
+
**Policy.** To override any default, put a JSON file at `.autoguard.json` (per project) or `~/.autoguard/policy.json` (global):
|
|
128
|
+
|
|
129
|
+
```json
|
|
130
|
+
{
|
|
131
|
+
"block_risk": 3.5,
|
|
132
|
+
"escalate_risk": 2.5,
|
|
133
|
+
"escalate_sensitive": 0.7,
|
|
134
|
+
"on_error": "escalate",
|
|
135
|
+
"escalation_model": "anthropic/claude-haiku-4.5",
|
|
136
|
+
"spend_cap_usd": 2.5
|
|
137
|
+
}
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
All settings and their defaults are in [`autoguard/policy.py`](https://github.com/rchandnaWUSTL/auto-guard/blob/main/autoguard/policy.py).
|
|
141
|
+
|
|
142
|
+
**Log.** Every decision is appended to `~/.autoguard/decisions.jsonl`. To log somewhere else, set `AUTOGUARD_LOG`. Once the logged spend reaches `spend_cap_usd`, escalations stop calling the LLM.
|
|
143
|
+
|
|
144
|
+
---
|
|
145
|
+
|
|
146
|
+
## Commands
|
|
147
|
+
|
|
148
|
+
| Command | Does |
|
|
149
|
+
|---|---|
|
|
150
|
+
| `autoguard install [--user]` | Adds the Claude Code hook. Your existing settings are kept, and running it twice is safe. |
|
|
151
|
+
| `autoguard check` | Runs two sample calls to verify your key |
|
|
152
|
+
| `autoguard console` | Opens the live decision dashboard |
|
|
153
|
+
| `autoguard demo [block\|escalate\|all]` | Runs the scripted demo calls through the real guard |
|
|
154
|
+
| `autoguard eval` | Scores the policy on the labeled test calls (run it from a clone of this repo) |
|
|
155
|
+
|
|
156
|
+
---
|
|
157
|
+
|
|
158
|
+
## Results
|
|
159
|
+
|
|
160
|
+
Held-out test set of 20 calls, never used for tuning ([`evals/holdout.jsonl`](https://github.com/rchandnaWUSTL/auto-guard/blob/main/evals/holdout.jsonl)):
|
|
161
|
+
|
|
162
|
+
| | Safe allowed | Dangerous caught | Ambiguous escalated | Median latency | Cost per check |
|
|
163
|
+
|---|---|---|---|---|---|
|
|
164
|
+
| **Auto-Guard (Jev)** | **8/8** | **8/8** | **4/4** | **428ms** | **$0.000025** |
|
|
165
|
+
| Claude Haiku 4.5 as the guard | 8/8 | 8/8 | 3/4 | 823ms | $0.00012 |
|
|
166
|
+
| Claude Sonnet 5 as the guard | 7/8 | 5/8 | 1/4 | 1,992ms | $0.00032 |
|
|
167
|
+
|
|
168
|
+
On the 61-call tuning set, it allowed 26/26 safe calls and caught 25/25 dangerous ones. The thresholds were tuned on that set.
|
|
169
|
+
|
|
170
|
+
Latency was measured from a laptop via OpenRouter, with a new connection per call (the same as the Claude Code hook). Calling `guard()` repeatedly in one process reuses the connection, which brings the median to about 310–360ms.
|
|
171
|
+
|
|
172
|
+
---
|
|
173
|
+
|
|
174
|
+
## Limits
|
|
175
|
+
|
|
176
|
+
- **It's a filter, not a sandbox.** It will sometimes miss things. Keep backups, scoped credentials and sandboxing.
|
|
177
|
+
- **"In scope" is its weakest signal.** Jev sometimes misjudges multi-part requests. So scope only counts when the risk is also high.
|
|
178
|
+
- **It can block things you asked for.** "Delete my old AWS config" gets blocked because deleting credentials is high-risk. Run those commands yourself.
|
|
179
|
+
- **Each check adds about 0.3–0.6s.** That goes unnoticed next to an agent's own model calls, but it isn't free.
|
|
180
|
+
|
|
181
|
+
---
|
|
182
|
+
|
|
183
|
+
## Development
|
|
184
|
+
|
|
185
|
+
```bash
|
|
186
|
+
python3 -m unittest discover tests # offline tests, no API calls
|
|
187
|
+
python3 -m autoguard eval # re-score using cached Jev answers (free)
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
The demo videos are generated from recorded, real Jev responses. The same inputs always produce identical files:
|
|
191
|
+
|
|
192
|
+
```bash
|
|
193
|
+
cd demo && npm install && npm run video # writes demo/out/*.mp4
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
```
|
|
197
|
+
autoguard/ guard(), policy, Jev client, Claude Code hook, CLI, live console
|
|
198
|
+
evals/ labeled test calls and cached Jev answers
|
|
199
|
+
tests/ unit tests
|
|
200
|
+
demo/ demo scenes, video capture (Playwright) and editing (Remotion)
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
Built on [TypeSafe's Jev](https://typesafe.ai).
|
|
204
|
+
|
|
205
|
+
## License
|
|
206
|
+
|
|
207
|
+
MIT. See [LICENSE](https://github.com/rchandnaWUSTL/auto-guard/blob/main/LICENSE).
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
agent_autoguard.egg-info/PKG-INFO
|
|
5
|
+
agent_autoguard.egg-info/SOURCES.txt
|
|
6
|
+
agent_autoguard.egg-info/dependency_links.txt
|
|
7
|
+
agent_autoguard.egg-info/entry_points.txt
|
|
8
|
+
agent_autoguard.egg-info/top_level.txt
|
|
9
|
+
autoguard/__init__.py
|
|
10
|
+
autoguard/__main__.py
|
|
11
|
+
autoguard/cli.py
|
|
12
|
+
autoguard/demo.py
|
|
13
|
+
autoguard/escalate.py
|
|
14
|
+
autoguard/evals.py
|
|
15
|
+
autoguard/guard.py
|
|
16
|
+
autoguard/jev.py
|
|
17
|
+
autoguard/log.py
|
|
18
|
+
autoguard/policy.py
|
|
19
|
+
autoguard/console/__init__.py
|
|
20
|
+
autoguard/console/index.html
|
|
21
|
+
autoguard/console/server.py
|
|
22
|
+
autoguard/hooks/__init__.py
|
|
23
|
+
autoguard/hooks/claude_code.py
|
|
24
|
+
tests/test_autoguard.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
autoguard
|