trifectaguard 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- trifectaguard-0.1.0/LICENSE +21 -0
- trifectaguard-0.1.0/PKG-INFO +238 -0
- trifectaguard-0.1.0/README.md +218 -0
- trifectaguard-0.1.0/pyproject.toml +33 -0
- trifectaguard-0.1.0/setup.cfg +4 -0
- trifectaguard-0.1.0/src/gateway/__init__.py +16 -0
- trifectaguard-0.1.0/src/gateway/__main__.py +167 -0
- trifectaguard-0.1.0/src/gateway/adjudicator.py +91 -0
- trifectaguard-0.1.0/src/gateway/config.py +59 -0
- trifectaguard-0.1.0/src/gateway/detector.py +74 -0
- trifectaguard-0.1.0/src/gateway/dlp.py +82 -0
- trifectaguard-0.1.0/src/gateway/engine.py +318 -0
- trifectaguard-0.1.0/src/gateway/gateway.py +276 -0
- trifectaguard-0.1.0/src/gateway/guard.py +165 -0
- trifectaguard-0.1.0/src/gateway/hook.py +235 -0
- trifectaguard-0.1.0/src/gateway/pins.py +54 -0
- trifectaguard-0.1.0/src/gateway/policies/claude-code.yaml +99 -0
- trifectaguard-0.1.0/src/gateway/policies/fetch.yaml +17 -0
- trifectaguard-0.1.0/src/gateway/policies/filesystem.yaml +57 -0
- trifectaguard-0.1.0/src/gateway/policies/github.yaml +63 -0
- trifectaguard-0.1.0/src/gateway/policies/mock.yaml +18 -0
- trifectaguard-0.1.0/src/gateway/policies/slack.yaml +23 -0
- trifectaguard-0.1.0/src/gateway/policy.py +160 -0
- trifectaguard-0.1.0/src/gateway/proxy.py +132 -0
- trifectaguard-0.1.0/src/gateway/rules.py +169 -0
- trifectaguard-0.1.0/src/gateway/scan.py +308 -0
- trifectaguard-0.1.0/src/gateway/suggest.py +71 -0
- trifectaguard-0.1.0/tests/test_flow_engine.py +227 -0
- trifectaguard-0.1.0/tests/test_gateway_e2e.py +206 -0
- trifectaguard-0.1.0/tests/test_guard.py +125 -0
- trifectaguard-0.1.0/tests/test_hook.py +198 -0
- trifectaguard-0.1.0/tests/test_properties.py +194 -0
- trifectaguard-0.1.0/tests/test_scan.py +122 -0
- trifectaguard-0.1.0/tests/test_taint.py +55 -0
- trifectaguard-0.1.0/trifectaguard.egg-info/PKG-INFO +238 -0
- trifectaguard-0.1.0/trifectaguard.egg-info/SOURCES.txt +38 -0
- trifectaguard-0.1.0/trifectaguard.egg-info/dependency_links.txt +1 -0
- trifectaguard-0.1.0/trifectaguard.egg-info/entry_points.txt +2 -0
- trifectaguard-0.1.0/trifectaguard.egg-info/requires.txt +9 -0
- trifectaguard-0.1.0/trifectaguard.egg-info/top_level.txt +1 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Ekaansh Jain
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,238 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: trifectaguard
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Information-flow control for AI agents: stop tool calls that move untrusted-influenced data or actions somewhere they shouldn't go. Claude Code hooks, an MCP proxy, or a library.
|
|
5
|
+
License: MIT
|
|
6
|
+
Project-URL: Homepage, https://github.com/Ekaansh-Jain/trifectaguard
|
|
7
|
+
Project-URL: Issues, https://github.com/Ekaansh-Jain/trifectaguard/issues
|
|
8
|
+
Project-URL: Security, https://github.com/Ekaansh-Jain/trifectaguard/security
|
|
9
|
+
Requires-Python: >=3.10
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Requires-Dist: pyyaml>=6
|
|
13
|
+
Provides-Extra: mcp
|
|
14
|
+
Requires-Dist: mcp<3,>=1.26; extra == "mcp"
|
|
15
|
+
Requires-Dist: anyio>=4; extra == "mcp"
|
|
16
|
+
Provides-Extra: detector
|
|
17
|
+
Requires-Dist: torch; extra == "detector"
|
|
18
|
+
Requires-Dist: transformers; extra == "detector"
|
|
19
|
+
Dynamic: license-file
|
|
20
|
+
|
|
21
|
+
# trifectaguard
|
|
22
|
+
|
|
23
|
+
**Stop AI agents from leaking your data or acting for an attacker, whatever
|
|
24
|
+
the injected instruction says.**
|
|
25
|
+
|
|
26
|
+
An agent that can read untrusted content (web pages, issues, emails), read your
|
|
27
|
+
private data, and send things out can be turned against you by one injected
|
|
28
|
+
instruction ([the "lethal trifecta"](https://simonwillison.net/tags/lethal-trifecta/)).
|
|
29
|
+
Detectors try to spot the injection's wording and can be evaded. trifectaguard
|
|
30
|
+
tracks **what the session has read** and **where each tool call sends it**, and
|
|
31
|
+
asks or blocks before private data, credentials, money or access go somewhere
|
|
32
|
+
an injection chose.
|
|
33
|
+
|
|
34
|
+
```
|
|
35
|
+
untrusted input → reads private data / credentials → sends it out ✗ blocked
|
|
36
|
+
(issue, web page, email) (post, email, curl, fetch, PR)
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
Works as **Claude Code hooks**, as a **proxy in front of any MCP server** (Claude
|
|
40
|
+
Desktop, Cursor, …), or as a **Python library** for your own agents
|
|
41
|
+
(LangChain/LangGraph, OpenAI Agents SDK, plain loops).
|
|
42
|
+
|
|
43
|
+
## Quick start
|
|
44
|
+
|
|
45
|
+
```bash
|
|
46
|
+
git clone https://github.com/Ekaansh-Jain/trifectaguard && cd trifectaguard
|
|
47
|
+
pip install ".[mcp]" # Python 3.10+; the [mcp] extra is only needed for the proxy
|
|
48
|
+
trifectaguard scan # read-only: what could an injection make your AI apps do?
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
`trifectaguard scan` reads the MCP configs of Claude Code, Claude Desktop, Cursor,
|
|
52
|
+
Windsurf and VS Code and explains every risky combination:
|
|
53
|
+
|
|
54
|
+
```
|
|
55
|
+
Claude Code (not protected)
|
|
56
|
+
HIGH An injected instruction could make the agent send your credentials out
|
|
57
|
+
untrusted input: WebFetch, WebSearch → then reads credentials: Read, Bash → then sends via: WebFetch, Bash
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
It changes nothing, starts no servers, never opens credential files and never
|
|
61
|
+
prints tokens from your configs. `--json` for tooling; it exits 1 when an
|
|
62
|
+
unprotected high-risk combination exists, so it can gate CI.
|
|
63
|
+
|
|
64
|
+
## Protect your agents
|
|
65
|
+
|
|
66
|
+
| | For | Verified in |
|
|
67
|
+
|---|---|---|
|
|
68
|
+
| [Claude Code hooks](#claude-code) | Claude Code, including `Read`, `WebFetch`, `Bash`, `Write` and every MCP server | a real Claude Code session |
|
|
69
|
+
| [MCP proxy](#any-mcp-app-claude-desktop-cursor-) | any MCP app; local (stdio) and remote (HTTP/SSE) servers | a real Claude Desktop chat |
|
|
70
|
+
| [Python library](#your-own-agent-python) | LangChain/LangGraph, OpenAI Agents SDK, your own loop | a live LangGraph agent |
|
|
71
|
+
|
|
72
|
+
Start any of them in `mode: monitor` to see what it *would* stop before it stops anything,
|
|
73
|
+
then `trifectaguard suggest -c <config>` proposes config (trusted sites, remembered
|
|
74
|
+
approvals) that removes repeat prompts without weakening the rules.
|
|
75
|
+
|
|
76
|
+
### Claude Code
|
|
77
|
+
|
|
78
|
+
```bash
|
|
79
|
+
cp hooks.example.yaml hooks.yaml # list your MCP servers; starts in monitor mode
|
|
80
|
+
trifectaguard hooks-snippet -c hooks.yaml # paste the output into ~/.claude/settings.json
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
The engine runs as `UserPromptSubmit` / `PreToolUse` / `PostToolUse` hooks. It
|
|
84
|
+
sees your request (so addresses and links you typed are trusted), tracks the
|
|
85
|
+
whole session across built-in tools and MCP servers, and asks through Claude
|
|
86
|
+
Code's normal approval prompt. It only ever answers "ask" or "deny", never
|
|
87
|
+
"allow", so it can't loosen your permission settings. `Bash` is classified from
|
|
88
|
+
its text (commands that can send data out or hide what they do get the
|
|
89
|
+
scrutiny); that can't be complete, so keep Claude Code's own Bash permissions
|
|
90
|
+
on, and for strong guarantees on shell commands use Claude Code's `/sandbox`
|
|
91
|
+
(OS-level network limits) alongside trifectaguard. With `remember_approvals:
|
|
92
|
+
project`, a destination you approve in a project isn't asked again there.
|
|
93
|
+
|
|
94
|
+
### Any MCP app (Claude Desktop, Cursor, …)
|
|
95
|
+
|
|
96
|
+
```bash
|
|
97
|
+
cp gateway.example.yaml gateway.yaml # your servers + which repos are public/private
|
|
98
|
+
trifectaguard inspect -c gateway.yaml # how every tool is classified
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
In the app's MCP config, replace your servers with one entry running
|
|
102
|
+
`trifectaguard run -c /abs/path/gateway.yaml`. One gateway fronts all of them, so a
|
|
103
|
+
web page read through one server and a file read through another are one flow.
|
|
104
|
+
Risky calls ask the user through MCP elicitation where the app supports it,
|
|
105
|
+
and are blocked otherwise. Works with `mcp` 1.26+ and 2.x.
|
|
106
|
+
|
|
107
|
+
### Your own agent (Python)
|
|
108
|
+
|
|
109
|
+
```python
|
|
110
|
+
from trifectaguard import Guard, ask_in_terminal
|
|
111
|
+
from langchain_core.tools import tool # or the OpenAI Agents SDK's @function_tool
|
|
112
|
+
|
|
113
|
+
guard = Guard({"tools": {
|
|
114
|
+
"read_inbox": {"reads": ["untrusted", "private"]},
|
|
115
|
+
"send_email": {"writes": "external", "destination": ["to"]},
|
|
116
|
+
}}, on_ask=ask_in_terminal) # default: anything needing approval is denied
|
|
117
|
+
|
|
118
|
+
guard.user_message(user_request) # destinations the user typed are trusted
|
|
119
|
+
|
|
120
|
+
@tool
|
|
121
|
+
@guard.tool
|
|
122
|
+
def send_email(to: str, body: str) -> str:
|
|
123
|
+
"""Send an email."""
|
|
124
|
+
...
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
A refused call doesn't run; the tool returns a short explanation the model can
|
|
128
|
+
relay (or raises `Blocked` with `blocked="raise"`). One `Guard` per conversation.
|
|
129
|
+
|
|
130
|
+
## How it works
|
|
131
|
+
|
|
132
|
+
- **Labels.** Tool results add labels to the session: `untrusted` (issues, web
|
|
133
|
+
pages, chat), `private` (private repos, local files), `secret` (credentials).
|
|
134
|
+
- **Flow rules.** Before a call that sends something: untrusted + credentials →
|
|
135
|
+
outside is blocked; untrusted + private → public or outside asks; untrusted
|
|
136
|
+
content steering writes to CI/shell/agent config, access changes or deletions
|
|
137
|
+
asks.
|
|
138
|
+
- **Destination provenance.** For a recipient, IBAN, URL or user, it checks
|
|
139
|
+
where the value came from: your request or trusted data (fine) vs only
|
|
140
|
+
untrusted content (an injection chose it: ask). Matching is by whole token,
|
|
141
|
+
and anything the agent itself wrote after reading untrusted content stays
|
|
142
|
+
untrusted.
|
|
143
|
+
- **DLP.** A secret the session read can't leave, even base64/hex/URL-encoded,
|
|
144
|
+
reversed or spaced out.
|
|
145
|
+
- **Rug-pull pins** (proxy). Tool definitions are pinned; a changed one is
|
|
146
|
+
quarantined until you re-pin.
|
|
147
|
+
- **Policies** are small YAML files. Presets ship for GitHub, filesystem, fetch,
|
|
148
|
+
Slack and Claude Code's built-in tools; unknown tools are treated as
|
|
149
|
+
untrusted with an unknown destination.
|
|
150
|
+
|
|
151
|
+
## Results
|
|
152
|
+
|
|
153
|
+
On [AgentDojo](https://github.com/ethz-spylab/agentdojo) v1.2.2 with a
|
|
154
|
+
worst-case agent that obeys every injection it reads:
|
|
155
|
+
|
|
156
|
+
| Attacker's goal | Still succeeds with trifectaguard |
|
|
157
|
+
|---|---|
|
|
158
|
+
| steal data (299 attacks) | **0.3%** |
|
|
159
|
+
| hijack payments, access or contact (153) | **0%** |
|
|
160
|
+
| deletions (39) | **0%** |
|
|
161
|
+
| steer the agent among legitimate options (100) | 60% (not what it controls) |
|
|
162
|
+
|
|
163
|
+
About 31% of benign tasks need one approval (44% as an MCP proxy, which can't
|
|
164
|
+
see your request). It passes an adaptive red-team suite written against its own
|
|
165
|
+
rules (20/20). With live models, attacks succeeded 0/12 times per model on
|
|
166
|
+
AgentDojo banking (7/12 and 6/12 unprotected) and 0/20 in a LangGraph agent
|
|
167
|
+
(11/20 unprotected), and it blocked exfiltration in real Claude Code and Claude
|
|
168
|
+
Desktop sessions. Prompt-injection detectors
|
|
169
|
+
(including this repo's own) hid clean data on 39–74% of benign tasks.
|
|
170
|
+
Method, disclosed post-hoc changes and every number: [RESULTS.md](RESULTS.md).
|
|
171
|
+
|
|
172
|
+
## Limits
|
|
173
|
+
|
|
174
|
+
- It controls **where data and access go**, not which legitimate option an
|
|
175
|
+
agent picks, what text it writes to a legitimate recipient, or its final answer.
|
|
176
|
+
- Security depends on the policies being right; a misclassified tool is a hole.
|
|
177
|
+
`trifectaguard inspect` and `trifectaguard scan` show what each tool is treated as.
|
|
178
|
+
- `Bash` classification in Claude Code is pattern-based: pair it with Claude
|
|
179
|
+
Code's `/sandbox` for shell commands.
|
|
180
|
+
- GitHub repo visibility comes from your config, not the API.
|
|
181
|
+
- The proxy forwards tools (not MCP resources or prompts).
|
|
182
|
+
- Nobody outside this project has tried to break it yet: please do
|
|
183
|
+
([SECURITY.md](SECURITY.md)).
|
|
184
|
+
|
|
185
|
+
## Related projects
|
|
186
|
+
|
|
187
|
+
Other open-source guards for Claude Code, described from their own READMEs and
|
|
188
|
+
source at the commit read (2026-09-28). **Not measured side by side**: this is
|
|
189
|
+
how each is documented to work, not a benchmark.
|
|
190
|
+
|
|
191
|
+
| | Acts on | Blocks or warns | Session memory | Notes from their docs/source |
|
|
192
|
+
|---|---|---|---|---|
|
|
193
|
+
| [lasso-security/claude-hooks](https://github.com/lasso-security/claude-hooks) (`8fbfd14`) | tool **output** (PostToolUse) | warns only | no | ~96 regex patterns in 4 categories (instruction override, role-play, encoding, context manipulation) |
|
|
194
|
+
| [dwarvesf/claude-guardrails](https://github.com/dwarvesf/claude-guardrails) (`b3c3e15`) | tool calls (PreToolUse), prompts, output | blocks (deny rules, command checks); output scan warns | no | mostly Claude Code permission deny rules for sensitive paths; its README notes Bash reads aren't covered by `Read` deny rules. Its output scanner reads a `tool_output` field; Claude Code's documented field is `tool_response` |
|
|
195
|
+
| [slavaspitsyn/claude-code-security-hooks](https://github.com/slavaspitsyn/claude-code-security-hooks) (`c4f126a`) | tool calls (PreToolUse) | blocks | no | per-command rules: credential path + network tool in the same command, read guards for credential directories, POST domain whitelist, canary files |
|
|
196
|
+
| **trifectaguard** | tool calls **and** output, across the whole session | asks or blocks | **yes** | labels what the session has read and checks where each call sends data and who chose the destination; also an MCP proxy and a Python library |
|
|
197
|
+
|
|
198
|
+
The others judge each call or output on its own (patterns, paths, command
|
|
199
|
+
shapes); trifectaguard judges a call by what the session has already read and where
|
|
200
|
+
the call sends it. They are complementary: path deny rules and trifectaguard can
|
|
201
|
+
run together.
|
|
202
|
+
|
|
203
|
+
## Research and reproducing the results
|
|
204
|
+
|
|
205
|
+
This repo started as a prompt-injection lab: a leak-rate benchmark of models
|
|
206
|
+
against documented 2026 agent attacks (mock MCP server, fake canary secret),
|
|
207
|
+
and a benchmark of injection detectors on tool outputs, including a fine-tuned
|
|
208
|
+
ModernBERT (`scripts/`, `data/`, results and corrections in RESULTS.md).
|
|
209
|
+
|
|
210
|
+
```bash
|
|
211
|
+
pip install ".[mcp]" agentdojo openai python-dotenv
|
|
212
|
+
cp .env.example .env # Groq / NVIDIA NIM / Gemini keys, for live-model runs only
|
|
213
|
+
python run_all_tests.py --no-llm # every test, no API calls
|
|
214
|
+
python eval/redteam/adaptive.py # attacks on trifectaguard's own rules
|
|
215
|
+
python eval/agentdojo/worst_case.py --hook # AgentDojo, model-independent (~3 min)
|
|
216
|
+
python eval/agentdojo/report.py # tables
|
|
217
|
+
python run_pilot.py --all --runs 10 # original model leak-rate benchmark
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
Inside this checkout the package is also importable as `src.gateway`
|
|
221
|
+
(`python -m src.gateway …`), which the research scripts use.
|
|
222
|
+
|
|
223
|
+
## Repository layout
|
|
224
|
+
|
|
225
|
+
- `src/gateway/`: the package (`trifectaguard` when installed): `engine.py` (labels +
|
|
226
|
+
flow rules), `rules.py` (YAML policies), `policies/` (presets), `dlp.py`,
|
|
227
|
+
`hook.py` (Claude Code), `gateway.py` (MCP proxy), `guard.py` (library),
|
|
228
|
+
`scan.py`, `pins.py`. `proxy.py`/`policy.py` are the original research proxy.
|
|
229
|
+
- `eval/`: AgentDojo, red-team, live-agent, Claude Code and Claude Desktop checks.
|
|
230
|
+
- `tests/`: unit and end-to-end tests.
|
|
231
|
+
- `src/servers/github_mock.py`, `attacks/`, `src/harness/`, `run_pilot.py`: the
|
|
232
|
+
research benchmark (sandbox server with a fake secret).
|
|
233
|
+
|
|
234
|
+
## Ethics
|
|
235
|
+
|
|
236
|
+
The benchmarks measure the vulnerability of *your own* agent setup in a sandbox
|
|
237
|
+
so a defense can block it. Attack payloads recreate the *structure* of public
|
|
238
|
+
disclosures and target only fake in-repo secrets.
|
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
# trifectaguard
|
|
2
|
+
|
|
3
|
+
**Stop AI agents from leaking your data or acting for an attacker, whatever
|
|
4
|
+
the injected instruction says.**
|
|
5
|
+
|
|
6
|
+
An agent that can read untrusted content (web pages, issues, emails), read your
|
|
7
|
+
private data, and send things out can be turned against you by one injected
|
|
8
|
+
instruction ([the "lethal trifecta"](https://simonwillison.net/tags/lethal-trifecta/)).
|
|
9
|
+
Detectors try to spot the injection's wording and can be evaded. trifectaguard
|
|
10
|
+
tracks **what the session has read** and **where each tool call sends it**, and
|
|
11
|
+
asks or blocks before private data, credentials, money or access go somewhere
|
|
12
|
+
an injection chose.
|
|
13
|
+
|
|
14
|
+
```
|
|
15
|
+
untrusted input → reads private data / credentials → sends it out ✗ blocked
|
|
16
|
+
(issue, web page, email) (post, email, curl, fetch, PR)
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
Works as **Claude Code hooks**, as a **proxy in front of any MCP server** (Claude
|
|
20
|
+
Desktop, Cursor, …), or as a **Python library** for your own agents
|
|
21
|
+
(LangChain/LangGraph, OpenAI Agents SDK, plain loops).
|
|
22
|
+
|
|
23
|
+
## Quick start
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
git clone https://github.com/Ekaansh-Jain/trifectaguard && cd trifectaguard
|
|
27
|
+
pip install ".[mcp]" # Python 3.10+; the [mcp] extra is only needed for the proxy
|
|
28
|
+
trifectaguard scan # read-only: what could an injection make your AI apps do?
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
`trifectaguard scan` reads the MCP configs of Claude Code, Claude Desktop, Cursor,
|
|
32
|
+
Windsurf and VS Code and explains every risky combination:
|
|
33
|
+
|
|
34
|
+
```
|
|
35
|
+
Claude Code (not protected)
|
|
36
|
+
HIGH An injected instruction could make the agent send your credentials out
|
|
37
|
+
untrusted input: WebFetch, WebSearch → then reads credentials: Read, Bash → then sends via: WebFetch, Bash
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
It changes nothing, starts no servers, never opens credential files and never
|
|
41
|
+
prints tokens from your configs. `--json` for tooling; it exits 1 when an
|
|
42
|
+
unprotected high-risk combination exists, so it can gate CI.
|
|
43
|
+
|
|
44
|
+
## Protect your agents
|
|
45
|
+
|
|
46
|
+
| | For | Verified in |
|
|
47
|
+
|---|---|---|
|
|
48
|
+
| [Claude Code hooks](#claude-code) | Claude Code, including `Read`, `WebFetch`, `Bash`, `Write` and every MCP server | a real Claude Code session |
|
|
49
|
+
| [MCP proxy](#any-mcp-app-claude-desktop-cursor-) | any MCP app; local (stdio) and remote (HTTP/SSE) servers | a real Claude Desktop chat |
|
|
50
|
+
| [Python library](#your-own-agent-python) | LangChain/LangGraph, OpenAI Agents SDK, your own loop | a live LangGraph agent |
|
|
51
|
+
|
|
52
|
+
Start any of them in `mode: monitor` to see what it *would* stop before it stops anything,
|
|
53
|
+
then `trifectaguard suggest -c <config>` proposes config (trusted sites, remembered
|
|
54
|
+
approvals) that removes repeat prompts without weakening the rules.
|
|
55
|
+
|
|
56
|
+
### Claude Code
|
|
57
|
+
|
|
58
|
+
```bash
|
|
59
|
+
cp hooks.example.yaml hooks.yaml # list your MCP servers; starts in monitor mode
|
|
60
|
+
trifectaguard hooks-snippet -c hooks.yaml # paste the output into ~/.claude/settings.json
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
The engine runs as `UserPromptSubmit` / `PreToolUse` / `PostToolUse` hooks. It
|
|
64
|
+
sees your request (so addresses and links you typed are trusted), tracks the
|
|
65
|
+
whole session across built-in tools and MCP servers, and asks through Claude
|
|
66
|
+
Code's normal approval prompt. It only ever answers "ask" or "deny", never
|
|
67
|
+
"allow", so it can't loosen your permission settings. `Bash` is classified from
|
|
68
|
+
its text (commands that can send data out or hide what they do get the
|
|
69
|
+
scrutiny); that can't be complete, so keep Claude Code's own Bash permissions
|
|
70
|
+
on, and for strong guarantees on shell commands use Claude Code's `/sandbox`
|
|
71
|
+
(OS-level network limits) alongside trifectaguard. With `remember_approvals:
|
|
72
|
+
project`, a destination you approve in a project isn't asked again there.
|
|
73
|
+
|
|
74
|
+
### Any MCP app (Claude Desktop, Cursor, …)
|
|
75
|
+
|
|
76
|
+
```bash
|
|
77
|
+
cp gateway.example.yaml gateway.yaml # your servers + which repos are public/private
|
|
78
|
+
trifectaguard inspect -c gateway.yaml # how every tool is classified
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
In the app's MCP config, replace your servers with one entry running
|
|
82
|
+
`trifectaguard run -c /abs/path/gateway.yaml`. One gateway fronts all of them, so a
|
|
83
|
+
web page read through one server and a file read through another are one flow.
|
|
84
|
+
Risky calls ask the user through MCP elicitation where the app supports it,
|
|
85
|
+
and are blocked otherwise. Works with `mcp` 1.26+ and 2.x.
|
|
86
|
+
|
|
87
|
+
### Your own agent (Python)
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
from trifectaguard import Guard, ask_in_terminal
|
|
91
|
+
from langchain_core.tools import tool # or the OpenAI Agents SDK's @function_tool
|
|
92
|
+
|
|
93
|
+
guard = Guard({"tools": {
|
|
94
|
+
"read_inbox": {"reads": ["untrusted", "private"]},
|
|
95
|
+
"send_email": {"writes": "external", "destination": ["to"]},
|
|
96
|
+
}}, on_ask=ask_in_terminal) # default: anything needing approval is denied
|
|
97
|
+
|
|
98
|
+
guard.user_message(user_request) # destinations the user typed are trusted
|
|
99
|
+
|
|
100
|
+
@tool
|
|
101
|
+
@guard.tool
|
|
102
|
+
def send_email(to: str, body: str) -> str:
|
|
103
|
+
"""Send an email."""
|
|
104
|
+
...
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
A refused call doesn't run; the tool returns a short explanation the model can
|
|
108
|
+
relay (or raises `Blocked` with `blocked="raise"`). One `Guard` per conversation.
|
|
109
|
+
|
|
110
|
+
## How it works
|
|
111
|
+
|
|
112
|
+
- **Labels.** Tool results add labels to the session: `untrusted` (issues, web
|
|
113
|
+
pages, chat), `private` (private repos, local files), `secret` (credentials).
|
|
114
|
+
- **Flow rules.** Before a call that sends something: untrusted + credentials →
|
|
115
|
+
outside is blocked; untrusted + private → public or outside asks; untrusted
|
|
116
|
+
content steering writes to CI/shell/agent config, access changes or deletions
|
|
117
|
+
asks.
|
|
118
|
+
- **Destination provenance.** For a recipient, IBAN, URL or user, it checks
|
|
119
|
+
where the value came from: your request or trusted data (fine) vs only
|
|
120
|
+
untrusted content (an injection chose it: ask). Matching is by whole token,
|
|
121
|
+
and anything the agent itself wrote after reading untrusted content stays
|
|
122
|
+
untrusted.
|
|
123
|
+
- **DLP.** A secret the session read can't leave, even base64/hex/URL-encoded,
|
|
124
|
+
reversed or spaced out.
|
|
125
|
+
- **Rug-pull pins** (proxy). Tool definitions are pinned; a changed one is
|
|
126
|
+
quarantined until you re-pin.
|
|
127
|
+
- **Policies** are small YAML files. Presets ship for GitHub, filesystem, fetch,
|
|
128
|
+
Slack and Claude Code's built-in tools; unknown tools are treated as
|
|
129
|
+
untrusted with an unknown destination.
|
|
130
|
+
|
|
131
|
+
## Results
|
|
132
|
+
|
|
133
|
+
On [AgentDojo](https://github.com/ethz-spylab/agentdojo) v1.2.2 with a
|
|
134
|
+
worst-case agent that obeys every injection it reads:
|
|
135
|
+
|
|
136
|
+
| Attacker's goal | Still succeeds with trifectaguard |
|
|
137
|
+
|---|---|
|
|
138
|
+
| steal data (299 attacks) | **0.3%** |
|
|
139
|
+
| hijack payments, access or contact (153) | **0%** |
|
|
140
|
+
| deletions (39) | **0%** |
|
|
141
|
+
| steer the agent among legitimate options (100) | 60% (not what it controls) |
|
|
142
|
+
|
|
143
|
+
About 31% of benign tasks need one approval (44% as an MCP proxy, which can't
|
|
144
|
+
see your request). It passes an adaptive red-team suite written against its own
|
|
145
|
+
rules (20/20). With live models, attacks succeeded 0/12 times per model on
|
|
146
|
+
AgentDojo banking (7/12 and 6/12 unprotected) and 0/20 in a LangGraph agent
|
|
147
|
+
(11/20 unprotected), and it blocked exfiltration in real Claude Code and Claude
|
|
148
|
+
Desktop sessions. Prompt-injection detectors
|
|
149
|
+
(including this repo's own) hid clean data on 39–74% of benign tasks.
|
|
150
|
+
Method, disclosed post-hoc changes and every number: [RESULTS.md](RESULTS.md).
|
|
151
|
+
|
|
152
|
+
## Limits
|
|
153
|
+
|
|
154
|
+
- It controls **where data and access go**, not which legitimate option an
|
|
155
|
+
agent picks, what text it writes to a legitimate recipient, or its final answer.
|
|
156
|
+
- Security depends on the policies being right; a misclassified tool is a hole.
|
|
157
|
+
`trifectaguard inspect` and `trifectaguard scan` show what each tool is treated as.
|
|
158
|
+
- `Bash` classification in Claude Code is pattern-based: pair it with Claude
|
|
159
|
+
Code's `/sandbox` for shell commands.
|
|
160
|
+
- GitHub repo visibility comes from your config, not the API.
|
|
161
|
+
- The proxy forwards tools (not MCP resources or prompts).
|
|
162
|
+
- Nobody outside this project has tried to break it yet: please do
|
|
163
|
+
([SECURITY.md](SECURITY.md)).
|
|
164
|
+
|
|
165
|
+
## Related projects
|
|
166
|
+
|
|
167
|
+
Other open-source guards for Claude Code, described from their own READMEs and
|
|
168
|
+
source at the commit read (2026-09-28). **Not measured side by side**: this is
|
|
169
|
+
how each is documented to work, not a benchmark.
|
|
170
|
+
|
|
171
|
+
| | Acts on | Blocks or warns | Session memory | Notes from their docs/source |
|
|
172
|
+
|---|---|---|---|---|
|
|
173
|
+
| [lasso-security/claude-hooks](https://github.com/lasso-security/claude-hooks) (`8fbfd14`) | tool **output** (PostToolUse) | warns only | no | ~96 regex patterns in 4 categories (instruction override, role-play, encoding, context manipulation) |
|
|
174
|
+
| [dwarvesf/claude-guardrails](https://github.com/dwarvesf/claude-guardrails) (`b3c3e15`) | tool calls (PreToolUse), prompts, output | blocks (deny rules, command checks); output scan warns | no | mostly Claude Code permission deny rules for sensitive paths; its README notes Bash reads aren't covered by `Read` deny rules. Its output scanner reads a `tool_output` field; Claude Code's documented field is `tool_response` |
|
|
175
|
+
| [slavaspitsyn/claude-code-security-hooks](https://github.com/slavaspitsyn/claude-code-security-hooks) (`c4f126a`) | tool calls (PreToolUse) | blocks | no | per-command rules: credential path + network tool in the same command, read guards for credential directories, POST domain whitelist, canary files |
|
|
176
|
+
| **trifectaguard** | tool calls **and** output, across the whole session | asks or blocks | **yes** | labels what the session has read and checks where each call sends data and who chose the destination; also an MCP proxy and a Python library |
|
|
177
|
+
|
|
178
|
+
The others judge each call or output on its own (patterns, paths, command
|
|
179
|
+
shapes); trifectaguard judges a call by what the session has already read and where
|
|
180
|
+
the call sends it. They are complementary: path deny rules and trifectaguard can
|
|
181
|
+
run together.
|
|
182
|
+
|
|
183
|
+
## Research and reproducing the results
|
|
184
|
+
|
|
185
|
+
This repo started as a prompt-injection lab: a leak-rate benchmark of models
|
|
186
|
+
against documented 2026 agent attacks (mock MCP server, fake canary secret),
|
|
187
|
+
and a benchmark of injection detectors on tool outputs, including a fine-tuned
|
|
188
|
+
ModernBERT (`scripts/`, `data/`, results and corrections in RESULTS.md).
|
|
189
|
+
|
|
190
|
+
```bash
|
|
191
|
+
pip install ".[mcp]" agentdojo openai python-dotenv
|
|
192
|
+
cp .env.example .env # Groq / NVIDIA NIM / Gemini keys, for live-model runs only
|
|
193
|
+
python run_all_tests.py --no-llm # every test, no API calls
|
|
194
|
+
python eval/redteam/adaptive.py # attacks on trifectaguard's own rules
|
|
195
|
+
python eval/agentdojo/worst_case.py --hook # AgentDojo, model-independent (~3 min)
|
|
196
|
+
python eval/agentdojo/report.py # tables
|
|
197
|
+
python run_pilot.py --all --runs 10 # original model leak-rate benchmark
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
Inside this checkout the package is also importable as `src.gateway`
|
|
201
|
+
(`python -m src.gateway …`), which the research scripts use.
|
|
202
|
+
|
|
203
|
+
## Repository layout
|
|
204
|
+
|
|
205
|
+
- `src/gateway/`: the package (`trifectaguard` when installed): `engine.py` (labels +
|
|
206
|
+
flow rules), `rules.py` (YAML policies), `policies/` (presets), `dlp.py`,
|
|
207
|
+
`hook.py` (Claude Code), `gateway.py` (MCP proxy), `guard.py` (library),
|
|
208
|
+
`scan.py`, `pins.py`. `proxy.py`/`policy.py` are the original research proxy.
|
|
209
|
+
- `eval/`: AgentDojo, red-team, live-agent, Claude Code and Claude Desktop checks.
|
|
210
|
+
- `tests/`: unit and end-to-end tests.
|
|
211
|
+
- `src/servers/github_mock.py`, `attacks/`, `src/harness/`, `run_pilot.py`: the
|
|
212
|
+
research benchmark (sandbox server with a fake secret).
|
|
213
|
+
|
|
214
|
+
## Ethics
|
|
215
|
+
|
|
216
|
+
The benchmarks measure the vulnerability of *your own* agent setup in a sandbox
|
|
217
|
+
so a defense can block it. Attack payloads recreate the *structure* of public
|
|
218
|
+
disclosures and target only fake in-repo secrets.
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "trifectaguard"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Information-flow control for AI agents: stop tool calls that move untrusted-influenced data or actions somewhere they shouldn't go. Claude Code hooks, an MCP proxy, or a library."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = {text = "MIT"}
|
|
12
|
+
dependencies = ["pyyaml>=6"]
|
|
13
|
+
|
|
14
|
+
[project.urls]
|
|
15
|
+
Homepage = "https://github.com/Ekaansh-Jain/trifectaguard"
|
|
16
|
+
Issues = "https://github.com/Ekaansh-Jain/trifectaguard/issues"
|
|
17
|
+
Security = "https://github.com/Ekaansh-Jain/trifectaguard/security"
|
|
18
|
+
|
|
19
|
+
[project.optional-dependencies]
|
|
20
|
+
mcp = ["mcp>=1.26,<3", "anyio>=4"] # the MCP proxy; tested on mcp 1.26 and 2.2
|
|
21
|
+
detector = ["torch", "transformers"] # optional injection detector signal
|
|
22
|
+
|
|
23
|
+
[project.scripts]
|
|
24
|
+
trifectaguard = "trifectaguard.__main__:main"
|
|
25
|
+
|
|
26
|
+
# The code lives in src/gateway so the research scripts in this repo keep
|
|
27
|
+
# importing it as src.gateway; installed, it is the `trifectaguard` package.
|
|
28
|
+
[tool.setuptools]
|
|
29
|
+
package-dir = {"trifectaguard" = "src/gateway"}
|
|
30
|
+
packages = ["trifectaguard"]
|
|
31
|
+
|
|
32
|
+
[tool.setuptools.package-data]
|
|
33
|
+
trifectaguard = ["policies/*.yaml"]
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
"""Information-flow control for AI agents.
|
|
2
|
+
|
|
3
|
+
Three ways to run the same engine:
|
|
4
|
+
- library: `from trifectaguard import Guard` (see guard.py)
|
|
5
|
+
- Claude Code: hooks (`python -m trifectaguard hooks-snippet -c config.yaml`)
|
|
6
|
+
- any MCP app: a proxy in front of your MCP servers (`python -m trifectaguard run -c config.yaml`)
|
|
7
|
+
|
|
8
|
+
(In this repo the package is importable as `src.gateway`; proxy.py and
|
|
9
|
+
policy.py are the original single-server research proxy used by the
|
|
10
|
+
benchmarks.)
|
|
11
|
+
"""
|
|
12
|
+
from .engine import FlowEngine, Verdict
|
|
13
|
+
from .guard import ApprovalRequest, Blocked, Guard, ask_in_terminal
|
|
14
|
+
from .rules import ServerPolicy
|
|
15
|
+
|
|
16
|
+
__all__ = ["Guard", "Blocked", "ApprovalRequest", "ask_in_terminal", "FlowEngine", "Verdict", "ServerPolicy"]
|