trifectaguard 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. trifectaguard-0.1.0/LICENSE +21 -0
  2. trifectaguard-0.1.0/PKG-INFO +238 -0
  3. trifectaguard-0.1.0/README.md +218 -0
  4. trifectaguard-0.1.0/pyproject.toml +33 -0
  5. trifectaguard-0.1.0/setup.cfg +4 -0
  6. trifectaguard-0.1.0/src/gateway/__init__.py +16 -0
  7. trifectaguard-0.1.0/src/gateway/__main__.py +167 -0
  8. trifectaguard-0.1.0/src/gateway/adjudicator.py +91 -0
  9. trifectaguard-0.1.0/src/gateway/config.py +59 -0
  10. trifectaguard-0.1.0/src/gateway/detector.py +74 -0
  11. trifectaguard-0.1.0/src/gateway/dlp.py +82 -0
  12. trifectaguard-0.1.0/src/gateway/engine.py +318 -0
  13. trifectaguard-0.1.0/src/gateway/gateway.py +276 -0
  14. trifectaguard-0.1.0/src/gateway/guard.py +165 -0
  15. trifectaguard-0.1.0/src/gateway/hook.py +235 -0
  16. trifectaguard-0.1.0/src/gateway/pins.py +54 -0
  17. trifectaguard-0.1.0/src/gateway/policies/claude-code.yaml +99 -0
  18. trifectaguard-0.1.0/src/gateway/policies/fetch.yaml +17 -0
  19. trifectaguard-0.1.0/src/gateway/policies/filesystem.yaml +57 -0
  20. trifectaguard-0.1.0/src/gateway/policies/github.yaml +63 -0
  21. trifectaguard-0.1.0/src/gateway/policies/mock.yaml +18 -0
  22. trifectaguard-0.1.0/src/gateway/policies/slack.yaml +23 -0
  23. trifectaguard-0.1.0/src/gateway/policy.py +160 -0
  24. trifectaguard-0.1.0/src/gateway/proxy.py +132 -0
  25. trifectaguard-0.1.0/src/gateway/rules.py +169 -0
  26. trifectaguard-0.1.0/src/gateway/scan.py +308 -0
  27. trifectaguard-0.1.0/src/gateway/suggest.py +71 -0
  28. trifectaguard-0.1.0/tests/test_flow_engine.py +227 -0
  29. trifectaguard-0.1.0/tests/test_gateway_e2e.py +206 -0
  30. trifectaguard-0.1.0/tests/test_guard.py +125 -0
  31. trifectaguard-0.1.0/tests/test_hook.py +198 -0
  32. trifectaguard-0.1.0/tests/test_properties.py +194 -0
  33. trifectaguard-0.1.0/tests/test_scan.py +122 -0
  34. trifectaguard-0.1.0/tests/test_taint.py +55 -0
  35. trifectaguard-0.1.0/trifectaguard.egg-info/PKG-INFO +238 -0
  36. trifectaguard-0.1.0/trifectaguard.egg-info/SOURCES.txt +38 -0
  37. trifectaguard-0.1.0/trifectaguard.egg-info/dependency_links.txt +1 -0
  38. trifectaguard-0.1.0/trifectaguard.egg-info/entry_points.txt +2 -0
  39. trifectaguard-0.1.0/trifectaguard.egg-info/requires.txt +9 -0
  40. trifectaguard-0.1.0/trifectaguard.egg-info/top_level.txt +1 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Ekaansh Jain
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,238 @@
1
+ Metadata-Version: 2.4
2
+ Name: trifectaguard
3
+ Version: 0.1.0
4
+ Summary: Information-flow control for AI agents: stop tool calls that move untrusted-influenced data or actions somewhere they shouldn't go. Claude Code hooks, an MCP proxy, or a library.
5
+ License: MIT
6
+ Project-URL: Homepage, https://github.com/Ekaansh-Jain/trifectaguard
7
+ Project-URL: Issues, https://github.com/Ekaansh-Jain/trifectaguard/issues
8
+ Project-URL: Security, https://github.com/Ekaansh-Jain/trifectaguard/security
9
+ Requires-Python: >=3.10
10
+ Description-Content-Type: text/markdown
11
+ License-File: LICENSE
12
+ Requires-Dist: pyyaml>=6
13
+ Provides-Extra: mcp
14
+ Requires-Dist: mcp<3,>=1.26; extra == "mcp"
15
+ Requires-Dist: anyio>=4; extra == "mcp"
16
+ Provides-Extra: detector
17
+ Requires-Dist: torch; extra == "detector"
18
+ Requires-Dist: transformers; extra == "detector"
19
+ Dynamic: license-file
20
+
21
+ # trifectaguard
22
+
23
+ **Stop AI agents from leaking your data or acting for an attacker, whatever
24
+ the injected instruction says.**
25
+
26
+ An agent that can read untrusted content (web pages, issues, emails), read your
27
+ private data, and send things out can be turned against you by one injected
28
+ instruction ([the "lethal trifecta"](https://simonwillison.net/tags/lethal-trifecta/)).
29
+ Detectors try to spot the injection's wording and can be evaded. trifectaguard
30
+ tracks **what the session has read** and **where each tool call sends it**, and
31
+ asks or blocks before private data, credentials, money or access go somewhere
32
+ an injection chose.
33
+
34
+ ```
35
+ untrusted input → reads private data / credentials → sends it out ✗ blocked
36
+ (issue, web page, email) (post, email, curl, fetch, PR)
37
+ ```
38
+
39
+ Works as **Claude Code hooks**, as a **proxy in front of any MCP server** (Claude
40
+ Desktop, Cursor, …), or as a **Python library** for your own agents
41
+ (LangChain/LangGraph, OpenAI Agents SDK, plain loops).
42
+
43
+ ## Quick start
44
+
45
+ ```bash
46
+ git clone https://github.com/Ekaansh-Jain/trifectaguard && cd trifectaguard
47
+ pip install ".[mcp]" # Python 3.10+; the [mcp] extra is only needed for the proxy
48
+ trifectaguard scan # read-only: what could an injection make your AI apps do?
49
+ ```
50
+
51
+ `trifectaguard scan` reads the MCP configs of Claude Code, Claude Desktop, Cursor,
52
+ Windsurf and VS Code and explains every risky combination:
53
+
54
+ ```
55
+ Claude Code (not protected)
56
+ HIGH An injected instruction could make the agent send your credentials out
57
+ untrusted input: WebFetch, WebSearch → then reads credentials: Read, Bash → then sends via: WebFetch, Bash
58
+ ```
59
+
60
+ It changes nothing, starts no servers, never opens credential files and never
61
+ prints tokens from your configs. `--json` for tooling; it exits 1 when an
62
+ unprotected high-risk combination exists, so it can gate CI.
63
+
64
+ ## Protect your agents
65
+
66
+ | | For | Verified in |
67
+ |---|---|---|
68
+ | [Claude Code hooks](#claude-code) | Claude Code, including `Read`, `WebFetch`, `Bash`, `Write` and every MCP server | a real Claude Code session |
69
+ | [MCP proxy](#any-mcp-app-claude-desktop-cursor-) | any MCP app; local (stdio) and remote (HTTP/SSE) servers | a real Claude Desktop chat |
70
+ | [Python library](#your-own-agent-python) | LangChain/LangGraph, OpenAI Agents SDK, your own loop | a live LangGraph agent |
71
+
72
+ Start any of them in `mode: monitor` to see what it *would* stop before it stops anything,
73
+ then `trifectaguard suggest -c <config>` proposes config (trusted sites, remembered
74
+ approvals) that removes repeat prompts without weakening the rules.
75
+
76
+ ### Claude Code
77
+
78
+ ```bash
79
+ cp hooks.example.yaml hooks.yaml # list your MCP servers; starts in monitor mode
80
+ trifectaguard hooks-snippet -c hooks.yaml # paste the output into ~/.claude/settings.json
81
+ ```
82
+
83
+ The engine runs as `UserPromptSubmit` / `PreToolUse` / `PostToolUse` hooks. It
84
+ sees your request (so addresses and links you typed are trusted), tracks the
85
+ whole session across built-in tools and MCP servers, and asks through Claude
86
+ Code's normal approval prompt. It only ever answers "ask" or "deny", never
87
+ "allow", so it can't loosen your permission settings. `Bash` is classified from
88
+ its text (commands that can send data out or hide what they do get the
89
+ scrutiny); that can't be complete, so keep Claude Code's own Bash permissions
90
+ on, and for strong guarantees on shell commands use Claude Code's `/sandbox`
91
+ (OS-level network limits) alongside trifectaguard. With `remember_approvals:
92
+ project`, a destination you approve in a project isn't asked again there.
93
+
94
+ ### Any MCP app (Claude Desktop, Cursor, …)
95
+
96
+ ```bash
97
+ cp gateway.example.yaml gateway.yaml # your servers + which repos are public/private
98
+ trifectaguard inspect -c gateway.yaml # how every tool is classified
99
+ ```
100
+
101
+ In the app's MCP config, replace your servers with one entry running
102
+ `trifectaguard run -c /abs/path/gateway.yaml`. One gateway fronts all of them, so a
103
+ web page read through one server and a file read through another are one flow.
104
+ Risky calls ask the user through MCP elicitation where the app supports it,
105
+ and are blocked otherwise. Works with `mcp` 1.26+ and 2.x.
106
+
107
+ ### Your own agent (Python)
108
+
109
+ ```python
110
+ from trifectaguard import Guard, ask_in_terminal
111
+ from langchain_core.tools import tool # or the OpenAI Agents SDK's @function_tool
112
+
113
+ guard = Guard({"tools": {
114
+ "read_inbox": {"reads": ["untrusted", "private"]},
115
+ "send_email": {"writes": "external", "destination": ["to"]},
116
+ }}, on_ask=ask_in_terminal) # default: anything needing approval is denied
117
+
118
+ guard.user_message(user_request) # destinations the user typed are trusted
119
+
120
+ @tool
121
+ @guard.tool
122
+ def send_email(to: str, body: str) -> str:
123
+ """Send an email."""
124
+ ...
125
+ ```
126
+
127
+ A refused call doesn't run; the tool returns a short explanation the model can
128
+ relay (or raises `Blocked` with `blocked="raise"`). One `Guard` per conversation.
129
+
130
+ ## How it works
131
+
132
+ - **Labels.** Tool results add labels to the session: `untrusted` (issues, web
133
+ pages, chat), `private` (private repos, local files), `secret` (credentials).
134
+ - **Flow rules.** Before a call that sends something: untrusted + credentials →
135
+ outside is blocked; untrusted + private → public or outside asks; untrusted
136
+ content steering writes to CI/shell/agent config, access changes or deletions
137
+ asks.
138
+ - **Destination provenance.** For a recipient, IBAN, URL or user, it checks
139
+ where the value came from: your request or trusted data (fine) vs only
140
+ untrusted content (an injection chose it: ask). Matching is by whole token,
141
+ and anything the agent itself wrote after reading untrusted content stays
142
+ untrusted.
143
+ - **DLP.** A secret the session read can't leave, even base64/hex/URL-encoded,
144
+ reversed or spaced out.
145
+ - **Rug-pull pins** (proxy). Tool definitions are pinned; a changed one is
146
+ quarantined until you re-pin.
147
+ - **Policies** are small YAML files. Presets ship for GitHub, filesystem, fetch,
148
+ Slack and Claude Code's built-in tools; unknown tools are treated as
149
+ untrusted with an unknown destination.
150
+
151
+ ## Results
152
+
153
+ On [AgentDojo](https://github.com/ethz-spylab/agentdojo) v1.2.2 with a
154
+ worst-case agent that obeys every injection it reads:
155
+
156
+ | Attacker's goal | Still succeeds with trifectaguard |
157
+ |---|---|
158
+ | steal data (299 attacks) | **0.3%** |
159
+ | hijack payments, access or contact (153) | **0%** |
160
+ | deletions (39) | **0%** |
161
+ | steer the agent among legitimate options (100) | 60% (not what it controls) |
162
+
163
+ About 31% of benign tasks need one approval (44% as an MCP proxy, which can't
164
+ see your request). It passes an adaptive red-team suite written against its own
165
+ rules (20/20). With live models, attacks succeeded 0/12 times per model on
166
+ AgentDojo banking (7/12 and 6/12 unprotected) and 0/20 in a LangGraph agent
167
+ (11/20 unprotected), and it blocked exfiltration in real Claude Code and Claude
168
+ Desktop sessions. Prompt-injection detectors
169
+ (including this repo's own) hid clean data on 39–74% of benign tasks.
170
+ Method, disclosed post-hoc changes and every number: [RESULTS.md](RESULTS.md).
171
+
172
+ ## Limits
173
+
174
+ - It controls **where data and access go**, not which legitimate option an
175
+ agent picks, what text it writes to a legitimate recipient, or its final answer.
176
+ - Security depends on the policies being right; a misclassified tool is a hole.
177
+ `trifectaguard inspect` and `trifectaguard scan` show what each tool is treated as.
178
+ - `Bash` classification in Claude Code is pattern-based: pair it with Claude
179
+ Code's `/sandbox` for shell commands.
180
+ - GitHub repo visibility comes from your config, not the API.
181
+ - The proxy forwards tools (not MCP resources or prompts).
182
+ - Nobody outside this project has tried to break it yet: please do
183
+ ([SECURITY.md](SECURITY.md)).
184
+
185
+ ## Related projects
186
+
187
+ Other open-source guards for Claude Code, described from their own READMEs and
188
+ source at the commit read (2026-09-28). **Not measured side by side**: this is
189
+ how each is documented to work, not a benchmark.
190
+
191
+ | | Acts on | Blocks or warns | Session memory | Notes from their docs/source |
192
+ |---|---|---|---|---|
193
+ | [lasso-security/claude-hooks](https://github.com/lasso-security/claude-hooks) (`8fbfd14`) | tool **output** (PostToolUse) | warns only | no | ~96 regex patterns in 4 categories (instruction override, role-play, encoding, context manipulation) |
194
+ | [dwarvesf/claude-guardrails](https://github.com/dwarvesf/claude-guardrails) (`b3c3e15`) | tool calls (PreToolUse), prompts, output | blocks (deny rules, command checks); output scan warns | no | mostly Claude Code permission deny rules for sensitive paths; its README notes Bash reads aren't covered by `Read` deny rules. Its output scanner reads a `tool_output` field; Claude Code's documented field is `tool_response` |
195
+ | [slavaspitsyn/claude-code-security-hooks](https://github.com/slavaspitsyn/claude-code-security-hooks) (`c4f126a`) | tool calls (PreToolUse) | blocks | no | per-command rules: credential path + network tool in the same command, read guards for credential directories, POST domain whitelist, canary files |
196
+ | **trifectaguard** | tool calls **and** output, across the whole session | asks or blocks | **yes** | labels what the session has read and checks where each call sends data and who chose the destination; also an MCP proxy and a Python library |
197
+
198
+ The others judge each call or output on its own (patterns, paths, command
199
+ shapes); trifectaguard judges a call by what the session has already read and where
200
+ the call sends it. They are complementary: path deny rules and trifectaguard can
201
+ run together.
202
+
203
+ ## Research and reproducing the results
204
+
205
+ This repo started as a prompt-injection lab: a leak-rate benchmark of models
206
+ against documented 2026 agent attacks (mock MCP server, fake canary secret),
207
+ and a benchmark of injection detectors on tool outputs, including a fine-tuned
208
+ ModernBERT (`scripts/`, `data/`, results and corrections in RESULTS.md).
209
+
210
+ ```bash
211
+ pip install ".[mcp]" agentdojo openai python-dotenv
212
+ cp .env.example .env # Groq / NVIDIA NIM / Gemini keys, for live-model runs only
213
+ python run_all_tests.py --no-llm # every test, no API calls
214
+ python eval/redteam/adaptive.py # attacks on trifectaguard's own rules
215
+ python eval/agentdojo/worst_case.py --hook # AgentDojo, model-independent (~3 min)
216
+ python eval/agentdojo/report.py # tables
217
+ python run_pilot.py --all --runs 10 # original model leak-rate benchmark
218
+ ```
219
+
220
+ Inside this checkout the package is also importable as `src.gateway`
221
+ (`python -m src.gateway …`), which the research scripts use.
222
+
223
+ ## Repository layout
224
+
225
+ - `src/gateway/`: the package (`trifectaguard` when installed): `engine.py` (labels +
226
+ flow rules), `rules.py` (YAML policies), `policies/` (presets), `dlp.py`,
227
+ `hook.py` (Claude Code), `gateway.py` (MCP proxy), `guard.py` (library),
228
+ `scan.py`, `pins.py`. `proxy.py`/`policy.py` are the original research proxy.
229
+ - `eval/`: AgentDojo, red-team, live-agent, Claude Code and Claude Desktop checks.
230
+ - `tests/`: unit and end-to-end tests.
231
+ - `src/servers/github_mock.py`, `attacks/`, `src/harness/`, `run_pilot.py`: the
232
+ research benchmark (sandbox server with a fake secret).
233
+
234
+ ## Ethics
235
+
236
+ The benchmarks measure the vulnerability of *your own* agent setup in a sandbox
237
+ so a defense can block it. Attack payloads recreate the *structure* of public
238
+ disclosures and target only fake in-repo secrets.
@@ -0,0 +1,218 @@
1
+ # trifectaguard
2
+
3
+ **Stop AI agents from leaking your data or acting for an attacker, whatever
4
+ the injected instruction says.**
5
+
6
+ An agent that can read untrusted content (web pages, issues, emails), read your
7
+ private data, and send things out can be turned against you by one injected
8
+ instruction ([the "lethal trifecta"](https://simonwillison.net/tags/lethal-trifecta/)).
9
+ Detectors try to spot the injection's wording and can be evaded. trifectaguard
10
+ tracks **what the session has read** and **where each tool call sends it**, and
11
+ asks or blocks before private data, credentials, money or access go somewhere
12
+ an injection chose.
13
+
14
+ ```
15
+ untrusted input → reads private data / credentials → sends it out ✗ blocked
16
+ (issue, web page, email) (post, email, curl, fetch, PR)
17
+ ```
18
+
19
+ Works as **Claude Code hooks**, as a **proxy in front of any MCP server** (Claude
20
+ Desktop, Cursor, …), or as a **Python library** for your own agents
21
+ (LangChain/LangGraph, OpenAI Agents SDK, plain loops).
22
+
23
+ ## Quick start
24
+
25
+ ```bash
26
+ git clone https://github.com/Ekaansh-Jain/trifectaguard && cd trifectaguard
27
+ pip install ".[mcp]" # Python 3.10+; the [mcp] extra is only needed for the proxy
28
+ trifectaguard scan # read-only: what could an injection make your AI apps do?
29
+ ```
30
+
31
+ `trifectaguard scan` reads the MCP configs of Claude Code, Claude Desktop, Cursor,
32
+ Windsurf and VS Code and explains every risky combination:
33
+
34
+ ```
35
+ Claude Code (not protected)
36
+ HIGH An injected instruction could make the agent send your credentials out
37
+ untrusted input: WebFetch, WebSearch → then reads credentials: Read, Bash → then sends via: WebFetch, Bash
38
+ ```
39
+
40
+ It changes nothing, starts no servers, never opens credential files and never
41
+ prints tokens from your configs. `--json` for tooling; it exits 1 when an
42
+ unprotected high-risk combination exists, so it can gate CI.
43
+
44
+ ## Protect your agents
45
+
46
+ | | For | Verified in |
47
+ |---|---|---|
48
+ | [Claude Code hooks](#claude-code) | Claude Code, including `Read`, `WebFetch`, `Bash`, `Write` and every MCP server | a real Claude Code session |
49
+ | [MCP proxy](#any-mcp-app-claude-desktop-cursor-) | any MCP app; local (stdio) and remote (HTTP/SSE) servers | a real Claude Desktop chat |
50
+ | [Python library](#your-own-agent-python) | LangChain/LangGraph, OpenAI Agents SDK, your own loop | a live LangGraph agent |
51
+
52
+ Start any of them in `mode: monitor` to see what it *would* stop before it stops anything,
53
+ then `trifectaguard suggest -c <config>` proposes config (trusted sites, remembered
54
+ approvals) that removes repeat prompts without weakening the rules.
55
+
56
+ ### Claude Code
57
+
58
+ ```bash
59
+ cp hooks.example.yaml hooks.yaml # list your MCP servers; starts in monitor mode
60
+ trifectaguard hooks-snippet -c hooks.yaml # paste the output into ~/.claude/settings.json
61
+ ```
62
+
63
+ The engine runs as `UserPromptSubmit` / `PreToolUse` / `PostToolUse` hooks. It
64
+ sees your request (so addresses and links you typed are trusted), tracks the
65
+ whole session across built-in tools and MCP servers, and asks through Claude
66
+ Code's normal approval prompt. It only ever answers "ask" or "deny", never
67
+ "allow", so it can't loosen your permission settings. `Bash` is classified from
68
+ its text (commands that can send data out or hide what they do get the
69
+ scrutiny); that can't be complete, so keep Claude Code's own Bash permissions
70
+ on, and for strong guarantees on shell commands use Claude Code's `/sandbox`
71
+ (OS-level network limits) alongside trifectaguard. With `remember_approvals:
72
+ project`, a destination you approve in a project isn't asked again there.
73
+
74
+ ### Any MCP app (Claude Desktop, Cursor, …)
75
+
76
+ ```bash
77
+ cp gateway.example.yaml gateway.yaml # your servers + which repos are public/private
78
+ trifectaguard inspect -c gateway.yaml # how every tool is classified
79
+ ```
80
+
81
+ In the app's MCP config, replace your servers with one entry running
82
+ `trifectaguard run -c /abs/path/gateway.yaml`. One gateway fronts all of them, so a
83
+ web page read through one server and a file read through another are one flow.
84
+ Risky calls ask the user through MCP elicitation where the app supports it,
85
+ and are blocked otherwise. Works with `mcp` 1.26+ and 2.x.
86
+
87
+ ### Your own agent (Python)
88
+
89
+ ```python
90
+ from trifectaguard import Guard, ask_in_terminal
91
+ from langchain_core.tools import tool # or the OpenAI Agents SDK's @function_tool
92
+
93
+ guard = Guard({"tools": {
94
+ "read_inbox": {"reads": ["untrusted", "private"]},
95
+ "send_email": {"writes": "external", "destination": ["to"]},
96
+ }}, on_ask=ask_in_terminal) # default: anything needing approval is denied
97
+
98
+ guard.user_message(user_request) # destinations the user typed are trusted
99
+
100
+ @tool
101
+ @guard.tool
102
+ def send_email(to: str, body: str) -> str:
103
+ """Send an email."""
104
+ ...
105
+ ```
106
+
107
+ A refused call doesn't run; the tool returns a short explanation the model can
108
+ relay (or raises `Blocked` with `blocked="raise"`). One `Guard` per conversation.
109
+
110
+ ## How it works
111
+
112
+ - **Labels.** Tool results add labels to the session: `untrusted` (issues, web
113
+ pages, chat), `private` (private repos, local files), `secret` (credentials).
114
+ - **Flow rules.** Before a call that sends something: untrusted + credentials →
115
+ outside is blocked; untrusted + private → public or outside asks; untrusted
116
+ content steering writes to CI/shell/agent config, access changes or deletions
117
+ asks.
118
+ - **Destination provenance.** For a recipient, IBAN, URL or user, it checks
119
+ where the value came from: your request or trusted data (fine) vs only
120
+ untrusted content (an injection chose it: ask). Matching is by whole token,
121
+ and anything the agent itself wrote after reading untrusted content stays
122
+ untrusted.
123
+ - **DLP.** A secret the session read can't leave, even base64/hex/URL-encoded,
124
+ reversed or spaced out.
125
+ - **Rug-pull pins** (proxy). Tool definitions are pinned; a changed one is
126
+ quarantined until you re-pin.
127
+ - **Policies** are small YAML files. Presets ship for GitHub, filesystem, fetch,
128
+ Slack and Claude Code's built-in tools; unknown tools are treated as
129
+ untrusted with an unknown destination.
130
+
131
+ ## Results
132
+
133
+ On [AgentDojo](https://github.com/ethz-spylab/agentdojo) v1.2.2 with a
134
+ worst-case agent that obeys every injection it reads:
135
+
136
+ | Attacker's goal | Still succeeds with trifectaguard |
137
+ |---|---|
138
+ | steal data (299 attacks) | **0.3%** |
139
+ | hijack payments, access or contact (153) | **0%** |
140
+ | deletions (39) | **0%** |
141
+ | steer the agent among legitimate options (100) | 60% (not what it controls) |
142
+
143
+ About 31% of benign tasks need one approval (44% as an MCP proxy, which can't
144
+ see your request). It passes an adaptive red-team suite written against its own
145
+ rules (20/20). With live models, attacks succeeded 0/12 times per model on
146
+ AgentDojo banking (7/12 and 6/12 unprotected) and 0/20 in a LangGraph agent
147
+ (11/20 unprotected), and it blocked exfiltration in real Claude Code and Claude
148
+ Desktop sessions. Prompt-injection detectors
149
+ (including this repo's own) hid clean data on 39–74% of benign tasks.
150
+ Method, disclosed post-hoc changes and every number: [RESULTS.md](RESULTS.md).
151
+
152
+ ## Limits
153
+
154
+ - It controls **where data and access go**, not which legitimate option an
155
+ agent picks, what text it writes to a legitimate recipient, or its final answer.
156
+ - Security depends on the policies being right; a misclassified tool is a hole.
157
+ `trifectaguard inspect` and `trifectaguard scan` show what each tool is treated as.
158
+ - `Bash` classification in Claude Code is pattern-based: pair it with Claude
159
+ Code's `/sandbox` for shell commands.
160
+ - GitHub repo visibility comes from your config, not the API.
161
+ - The proxy forwards tools (not MCP resources or prompts).
162
+ - Nobody outside this project has tried to break it yet: please do
163
+ ([SECURITY.md](SECURITY.md)).
164
+
165
+ ## Related projects
166
+
167
+ Other open-source guards for Claude Code, described from their own READMEs and
168
+ source at the commit read (2026-09-28). **Not measured side by side**: this is
169
+ how each is documented to work, not a benchmark.
170
+
171
+ | | Acts on | Blocks or warns | Session memory | Notes from their docs/source |
172
+ |---|---|---|---|---|
173
+ | [lasso-security/claude-hooks](https://github.com/lasso-security/claude-hooks) (`8fbfd14`) | tool **output** (PostToolUse) | warns only | no | ~96 regex patterns in 4 categories (instruction override, role-play, encoding, context manipulation) |
174
+ | [dwarvesf/claude-guardrails](https://github.com/dwarvesf/claude-guardrails) (`b3c3e15`) | tool calls (PreToolUse), prompts, output | blocks (deny rules, command checks); output scan warns | no | mostly Claude Code permission deny rules for sensitive paths; its README notes Bash reads aren't covered by `Read` deny rules. Its output scanner reads a `tool_output` field; Claude Code's documented field is `tool_response` |
175
+ | [slavaspitsyn/claude-code-security-hooks](https://github.com/slavaspitsyn/claude-code-security-hooks) (`c4f126a`) | tool calls (PreToolUse) | blocks | no | per-command rules: credential path + network tool in the same command, read guards for credential directories, POST domain whitelist, canary files |
176
+ | **trifectaguard** | tool calls **and** output, across the whole session | asks or blocks | **yes** | labels what the session has read and checks where each call sends data and who chose the destination; also an MCP proxy and a Python library |
177
+
178
+ The others judge each call or output on its own (patterns, paths, command
179
+ shapes); trifectaguard judges a call by what the session has already read and where
180
+ the call sends it. They are complementary: path deny rules and trifectaguard can
181
+ run together.
182
+
183
+ ## Research and reproducing the results
184
+
185
+ This repo started as a prompt-injection lab: a leak-rate benchmark of models
186
+ against documented 2026 agent attacks (mock MCP server, fake canary secret),
187
+ and a benchmark of injection detectors on tool outputs, including a fine-tuned
188
+ ModernBERT (`scripts/`, `data/`, results and corrections in RESULTS.md).
189
+
190
+ ```bash
191
+ pip install ".[mcp]" agentdojo openai python-dotenv
192
+ cp .env.example .env # Groq / NVIDIA NIM / Gemini keys, for live-model runs only
193
+ python run_all_tests.py --no-llm # every test, no API calls
194
+ python eval/redteam/adaptive.py # attacks on trifectaguard's own rules
195
+ python eval/agentdojo/worst_case.py --hook # AgentDojo, model-independent (~3 min)
196
+ python eval/agentdojo/report.py # tables
197
+ python run_pilot.py --all --runs 10 # original model leak-rate benchmark
198
+ ```
199
+
200
+ Inside this checkout the package is also importable as `src.gateway`
201
+ (`python -m src.gateway …`), which the research scripts use.
202
+
203
+ ## Repository layout
204
+
205
+ - `src/gateway/`: the package (`trifectaguard` when installed): `engine.py` (labels +
206
+ flow rules), `rules.py` (YAML policies), `policies/` (presets), `dlp.py`,
207
+ `hook.py` (Claude Code), `gateway.py` (MCP proxy), `guard.py` (library),
208
+ `scan.py`, `pins.py`. `proxy.py`/`policy.py` are the original research proxy.
209
+ - `eval/`: AgentDojo, red-team, live-agent, Claude Code and Claude Desktop checks.
210
+ - `tests/`: unit and end-to-end tests.
211
+ - `src/servers/github_mock.py`, `attacks/`, `src/harness/`, `run_pilot.py`: the
212
+ research benchmark (sandbox server with a fake secret).
213
+
214
+ ## Ethics
215
+
216
+ The benchmarks measure the vulnerability of *your own* agent setup in a sandbox
217
+ so a defense can block it. Attack payloads recreate the *structure* of public
218
+ disclosures and target only fake in-repo secrets.
@@ -0,0 +1,33 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "trifectaguard"
7
+ version = "0.1.0"
8
+ description = "Information-flow control for AI agents: stop tool calls that move untrusted-influenced data or actions somewhere they shouldn't go. Claude Code hooks, an MCP proxy, or a library."
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = {text = "MIT"}
12
+ dependencies = ["pyyaml>=6"]
13
+
14
+ [project.urls]
15
+ Homepage = "https://github.com/Ekaansh-Jain/trifectaguard"
16
+ Issues = "https://github.com/Ekaansh-Jain/trifectaguard/issues"
17
+ Security = "https://github.com/Ekaansh-Jain/trifectaguard/security"
18
+
19
+ [project.optional-dependencies]
20
+ mcp = ["mcp>=1.26,<3", "anyio>=4"] # the MCP proxy; tested on mcp 1.26 and 2.2
21
+ detector = ["torch", "transformers"] # optional injection detector signal
22
+
23
+ [project.scripts]
24
+ trifectaguard = "trifectaguard.__main__:main"
25
+
26
+ # The code lives in src/gateway so the research scripts in this repo keep
27
+ # importing it as src.gateway; installed, it is the `trifectaguard` package.
28
+ [tool.setuptools]
29
+ package-dir = {"trifectaguard" = "src/gateway"}
30
+ packages = ["trifectaguard"]
31
+
32
+ [tool.setuptools.package-data]
33
+ trifectaguard = ["policies/*.yaml"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,16 @@
1
+ """Information-flow control for AI agents.
2
+
3
+ Three ways to run the same engine:
4
+ - library: `from trifectaguard import Guard` (see guard.py)
5
+ - Claude Code: hooks (`python -m trifectaguard hooks-snippet -c config.yaml`)
6
+ - any MCP app: a proxy in front of your MCP servers (`python -m trifectaguard run -c config.yaml`)
7
+
8
+ (In this repo the package is importable as `src.gateway`; proxy.py and
9
+ policy.py are the original single-server research proxy used by the
10
+ benchmarks.)
11
+ """
12
+ from .engine import FlowEngine, Verdict
13
+ from .guard import ApprovalRequest, Blocked, Guard, ask_in_terminal
14
+ from .rules import ServerPolicy
15
+
16
+ __all__ = ["Guard", "Blocked", "ApprovalRequest", "ask_in_terminal", "FlowEngine", "Verdict", "ServerPolicy"]