fusion-safety 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fusion_safety-0.1.0/.claude-plugin/marketplace.json +21 -0
- fusion_safety-0.1.0/MANIFEST.in +20 -0
- fusion_safety-0.1.0/PKG-INFO +165 -0
- fusion_safety-0.1.0/README.md +124 -0
- fusion_safety-0.1.0/cassettes/data_exfiltration.v1.json +651 -0
- fusion_safety-0.1.0/cassettes/direct_prompt_injection.v1.json +838 -0
- fusion_safety-0.1.0/cassettes/excessive_agency.v1.json +662 -0
- fusion_safety-0.1.0/cassettes/system_prompt_leakage.v1.json +651 -0
- fusion_safety-0.1.0/crosswalk/owasp_crosswalk.v2025.yaml +52 -0
- fusion_safety-0.1.0/datasets/gold/data_exfiltration.v1.jsonl +23 -0
- fusion_safety-0.1.0/datasets/gold/direct_prompt_injection.v1.jsonl +28 -0
- fusion_safety-0.1.0/datasets/gold/excessive_agency.v1.jsonl +24 -0
- fusion_safety-0.1.0/datasets/gold/instruction_following.v1.jsonl +16 -0
- fusion_safety-0.1.0/datasets/gold/system_prompt_leakage.v1.jsonl +24 -0
- fusion_safety-0.1.0/datasets/guard_bench/cases.jsonl +78 -0
- fusion_safety-0.1.0/datasets/probes/data_exfiltration.v1.jsonl +18 -0
- fusion_safety-0.1.0/datasets/probes/direct_prompt_injection.v1.jsonl +24 -0
- fusion_safety-0.1.0/datasets/probes/excessive_agency.v1.jsonl +18 -0
- fusion_safety-0.1.0/datasets/probes/system_prompt_leakage.v1.jsonl +18 -0
- fusion_safety-0.1.0/datasets/quality_tests/instruction_following.v1.jsonl +10 -0
- fusion_safety-0.1.0/datasets/reference_agents/v1.jsonl +5 -0
- fusion_safety-0.1.0/evals/baseline.data_exfiltration.json +11 -0
- fusion_safety-0.1.0/evals/baseline.direct_prompt_injection.json +11 -0
- fusion_safety-0.1.0/evals/baseline.excessive_agency.json +11 -0
- fusion_safety-0.1.0/evals/baseline.system_prompt_leakage.json +11 -0
- fusion_safety-0.1.0/evals/guard_bench.baseline.json +34 -0
- fusion_safety-0.1.0/evals/policy.yaml +35 -0
- fusion_safety-0.1.0/evals/validation/v1/judge_eval_prob_qwen7b.json +296 -0
- fusion_safety-0.1.0/frontend/dist/assets/index-CW6TCZzY.css +1 -0
- fusion_safety-0.1.0/frontend/dist/assets/index-Cb3URv5L.js +76 -0
- fusion_safety-0.1.0/frontend/dist/index.html +28 -0
- fusion_safety-0.1.0/frontend/dist/logo.png +0 -0
- fusion_safety-0.1.0/fusion_first/__init__.py +8 -0
- fusion_safety-0.1.0/fusion_first/_data.py +63 -0
- fusion_safety-0.1.0/fusion_first/attacks/__init__.py +3 -0
- fusion_safety-0.1.0/fusion_first/attacks/agentic.py +80 -0
- fusion_safety-0.1.0/fusion_first/attacks/taxonomy.py +44 -0
- fusion_safety-0.1.0/fusion_first/attacks/templates.py +160 -0
- fusion_safety-0.1.0/fusion_first/backends/__init__.py +1 -0
- fusion_safety-0.1.0/fusion_first/backends/presets.py +22 -0
- fusion_safety-0.1.0/fusion_first/backends/resolve.py +469 -0
- fusion_safety-0.1.0/fusion_first/cli.py +733 -0
- fusion_safety-0.1.0/fusion_first/engine/__init__.py +5 -0
- fusion_safety-0.1.0/fusion_first/engine/before_after.py +33 -0
- fusion_safety-0.1.0/fusion_first/engine/fixes.py +129 -0
- fusion_safety-0.1.0/fusion_first/engine/guard_replay.py +131 -0
- fusion_safety-0.1.0/fusion_first/engine/report.py +184 -0
- fusion_safety-0.1.0/fusion_first/engine/report_html.py +303 -0
- fusion_safety-0.1.0/fusion_first/engine/user_scan.py +629 -0
- fusion_safety-0.1.0/fusion_first/errors.py +141 -0
- fusion_safety-0.1.0/fusion_first/goldset.py +133 -0
- fusion_safety-0.1.0/fusion_first/guardrail/__init__.py +32 -0
- fusion_safety-0.1.0/fusion_first/guardrail/benchmark.py +271 -0
- fusion_safety-0.1.0/fusion_first/guardrail/client.py +28 -0
- fusion_safety-0.1.0/fusion_first/guardrail/guard.py +832 -0
- fusion_safety-0.1.0/fusion_first/guardrail/policy.py +81 -0
- fusion_safety-0.1.0/fusion_first/guardrail/simulate.py +81 -0
- fusion_safety-0.1.0/fusion_first/guardrail/snippet.py +50 -0
- fusion_safety-0.1.0/fusion_first/integrations/__init__.py +37 -0
- fusion_safety-0.1.0/fusion_first/integrations/agent_tools.py +427 -0
- fusion_safety-0.1.0/fusion_first/integrations/claude_hooks.py +1296 -0
- fusion_safety-0.1.0/fusion_first/integrations/live.py +176 -0
- fusion_safety-0.1.0/fusion_first/integrations/mcp_server.py +271 -0
- fusion_safety-0.1.0/fusion_first/integrations/promptfoo.py +180 -0
- fusion_safety-0.1.0/fusion_first/integrations/run_tools.py +192 -0
- fusion_safety-0.1.0/fusion_first/judge/__init__.py +15 -0
- fusion_safety-0.1.0/fusion_first/judge/judge.py +356 -0
- fusion_safety-0.1.0/fusion_first/judge/prob_judge.py +390 -0
- fusion_safety-0.1.0/fusion_first/judge/rubric.py +276 -0
- fusion_safety-0.1.0/fusion_first/measure/__init__.py +7 -0
- fusion_safety-0.1.0/fusion_first/measure/harness.py +359 -0
- fusion_safety-0.1.0/fusion_first/model/__init__.py +18 -0
- fusion_safety-0.1.0/fusion_first/model/client.py +77 -0
- fusion_safety-0.1.0/fusion_first/model/models.yaml +26 -0
- fusion_safety-0.1.0/fusion_first/model/providers/__init__.py +4 -0
- fusion_safety-0.1.0/fusion_first/model/providers/_common.py +44 -0
- fusion_safety-0.1.0/fusion_first/model/providers/anthropic.py +122 -0
- fusion_safety-0.1.0/fusion_first/model/providers/claude_cli.py +464 -0
- fusion_safety-0.1.0/fusion_first/model/providers/command.py +83 -0
- fusion_safety-0.1.0/fusion_first/model/providers/demo_detectors.py +300 -0
- fusion_safety-0.1.0/fusion_first/model/providers/heuristic.py +179 -0
- fusion_safety-0.1.0/fusion_first/model/providers/ollama.py +225 -0
- fusion_safety-0.1.0/fusion_first/model/providers/openai.py +93 -0
- fusion_safety-0.1.0/fusion_first/model/providers/openai_compat.py +105 -0
- fusion_safety-0.1.0/fusion_first/model/registry.py +108 -0
- fusion_safety-0.1.0/fusion_first/model/replay.py +198 -0
- fusion_safety-0.1.0/fusion_first/offline.py +16 -0
- fusion_safety-0.1.0/fusion_first/recording.py +172 -0
- fusion_safety-0.1.0/fusion_first/runs/__init__.py +1 -0
- fusion_safety-0.1.0/fusion_first/runs/cli.py +290 -0
- fusion_safety-0.1.0/fusion_first/runs/engine.py +858 -0
- fusion_safety-0.1.0/fusion_first/runs/logs.py +434 -0
- fusion_safety-0.1.0/fusion_first/runs/oracle_check.py +258 -0
- fusion_safety-0.1.0/fusion_first/runs/service.py +464 -0
- fusion_safety-0.1.0/fusion_first/schemas.py +394 -0
- fusion_safety-0.1.0/fusion_first/security/__init__.py +7 -0
- fusion_safety-0.1.0/fusion_first/security/budget.py +88 -0
- fusion_safety-0.1.0/fusion_first/security/normalize.py +89 -0
- fusion_safety-0.1.0/fusion_first/security/redaction.py +95 -0
- fusion_safety-0.1.0/fusion_first/security/webhook.py +36 -0
- fusion_safety-0.1.0/fusion_first/stats/__init__.py +5 -0
- fusion_safety-0.1.0/fusion_first/stats/calibration.py +70 -0
- fusion_safety-0.1.0/fusion_first/stats/gate.py +294 -0
- fusion_safety-0.1.0/fusion_first/stats/metrics.py +198 -0
- fusion_safety-0.1.0/fusion_first/stats/paired.py +176 -0
- fusion_safety-0.1.0/fusion_first/targets.py +35 -0
- fusion_safety-0.1.0/fusion_first/validate/__init__.py +1 -0
- fusion_safety-0.1.0/fusion_first/validate/ifeval_lite.py +575 -0
- fusion_safety-0.1.0/fusion_first/validate/oracles.py +552 -0
- fusion_safety-0.1.0/fusion_first/validate/prereg.py +98 -0
- fusion_safety-0.1.0/fusion_first/web/__init__.py +2 -0
- fusion_safety-0.1.0/fusion_first/web/clients.py +119 -0
- fusion_safety-0.1.0/fusion_first/web/factory.py +283 -0
- fusion_safety-0.1.0/fusion_first/web/models.py +95 -0
- fusion_safety-0.1.0/fusion_first/web/persistence.py +36 -0
- fusion_safety-0.1.0/fusion_first/web/serve.py +192 -0
- fusion_safety-0.1.0/fusion_first/web/settings.py +53 -0
- fusion_safety-0.1.0/fusion_first/web/snippet.py +8 -0
- fusion_safety-0.1.0/fusion_first/web/sse.py +65 -0
- fusion_safety-0.1.0/fusion_safety.egg-info/PKG-INFO +165 -0
- fusion_safety-0.1.0/fusion_safety.egg-info/SOURCES.txt +137 -0
- fusion_safety-0.1.0/fusion_safety.egg-info/dependency_links.txt +1 -0
- fusion_safety-0.1.0/fusion_safety.egg-info/entry_points.txt +3 -0
- fusion_safety-0.1.0/fusion_safety.egg-info/requires.txt +29 -0
- fusion_safety-0.1.0/fusion_safety.egg-info/top_level.txt +1 -0
- fusion_safety-0.1.0/plugins/fusion/.claude-plugin/plugin.json +19 -0
- fusion_safety-0.1.0/plugins/fusion/.mcp.json +9 -0
- fusion_safety-0.1.0/plugins/fusion/agents/fusion-judge.md +47 -0
- fusion_safety-0.1.0/plugins/fusion/skills/audit/SKILL.md +65 -0
- fusion_safety-0.1.0/plugins/fusion/skills/grade/SKILL.md +26 -0
- fusion_safety-0.1.0/plugins/fusion/skills/grade-logs/SKILL.md +21 -0
- fusion_safety-0.1.0/plugins/fusion/skills/guard/SKILL.md +24 -0
- fusion_safety-0.1.0/plugins/fusion/skills/harden/SKILL.md +20 -0
- fusion_safety-0.1.0/plugins/fusion/skills/setup/SKILL.md +23 -0
- fusion_safety-0.1.0/plugins/fusion-guard/.claude-plugin/plugin.json +9 -0
- fusion_safety-0.1.0/plugins/fusion-guard/hooks/hooks.json +16 -0
- fusion_safety-0.1.0/pyproject.toml +102 -0
- fusion_safety-0.1.0/setup.cfg +4 -0
- fusion_safety-0.1.0/setup.py +147 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "fusion-first",
|
|
3
|
+
"owner": {
|
|
4
|
+
"name": "Fusion First"
|
|
5
|
+
},
|
|
6
|
+
"plugins": [
|
|
7
|
+
{
|
|
8
|
+
"name": "fusion",
|
|
9
|
+
"source": "./plugins/fusion",
|
|
10
|
+
"description": "Audit an agent's prompt for safety and quality with no API key; Claude Code grades, Fusion measures the grader."
|
|
11
|
+
},
|
|
12
|
+
{
|
|
13
|
+
"name": "fusion-guard",
|
|
14
|
+
"source": "./plugins/fusion-guard",
|
|
15
|
+
"description": "Opt-in: guard Claude Code's own tool calls against exfiltration, destruction and prompt-injected content (asks, never auto-allows)."
|
|
16
|
+
}
|
|
17
|
+
],
|
|
18
|
+
"metadata": {
|
|
19
|
+
"description": "Keyless safety and quality evaluation for AI agents."
|
|
20
|
+
}
|
|
21
|
+
}
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
# The sdist carries only what builds the wheel (owner, 2026-09-29: leave out anything unnecessary): the
|
|
2
|
+
# package (minus setup.py's DEV_ONLY_MODULES), pyproject/setup, README (the PyPI page), the runtime data
|
|
3
|
+
# (setup.py _RUNTIME_DATA), the Claude Code marketplace and the built web app. No docs, tests, examples or
|
|
4
|
+
# evidence.
|
|
5
|
+
include README.md pyproject.toml setup.py
|
|
6
|
+
include cassettes/*.json
|
|
7
|
+
include crosswalk/*.yaml
|
|
8
|
+
include datasets/gold/*.jsonl datasets/probes/*.jsonl datasets/guard_bench/cases.jsonl
|
|
9
|
+
include datasets/quality_tests/*.jsonl datasets/reference_agents/*.jsonl
|
|
10
|
+
include evals/policy.yaml evals/baseline.*.json evals/guard_bench.baseline.json
|
|
11
|
+
include evals/validation/v1/judge_eval_prob_qwen7b.json
|
|
12
|
+
include .claude-plugin/marketplace.json
|
|
13
|
+
graft plugins
|
|
14
|
+
exclude plugins/*/README.md
|
|
15
|
+
graft frontend/dist
|
|
16
|
+
exclude frontend/dist/og-image.png
|
|
17
|
+
# fusion_first/_bundled/ is a build artifact (populated by setup.py), not shipped in the sdist.
|
|
18
|
+
prune fusion_first/_bundled
|
|
19
|
+
prune tests
|
|
20
|
+
global-exclude __pycache__ *.pyc *.pyo
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: fusion-safety
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Measure and guard AI agents: safety and quality evaluation with a runtime guardrail
|
|
5
|
+
Author: Fusion First
|
|
6
|
+
License: Proprietary
|
|
7
|
+
Project-URL: Homepage, https://fusion-first-testing.com
|
|
8
|
+
Project-URL: Documentation, https://fusion-first-testing.com/use
|
|
9
|
+
Project-URL: Trust Report, https://fusion-first-testing.com/trust
|
|
10
|
+
Keywords: llm,ai-safety,agentic,owasp,prompt-injection,guardrail,red-team,mcp
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: License :: Other/Proprietary License
|
|
13
|
+
Classifier: Topic :: Security
|
|
14
|
+
Classifier: Topic :: Software Development :: Quality Assurance
|
|
15
|
+
Requires-Python: >=3.11
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
Requires-Dist: pydantic>=2.9
|
|
18
|
+
Requires-Dist: numpy>=1.26
|
|
19
|
+
Requires-Dist: pyyaml>=6.0
|
|
20
|
+
Provides-Extra: api
|
|
21
|
+
Requires-Dist: anthropic>=0.40; extra == "api"
|
|
22
|
+
Requires-Dist: openai>=1.50; extra == "api"
|
|
23
|
+
Provides-Extra: serve
|
|
24
|
+
Requires-Dist: fastapi>=0.115; extra == "serve"
|
|
25
|
+
Requires-Dist: uvicorn>=0.30; extra == "serve"
|
|
26
|
+
Requires-Dist: pydantic-settings>=2.4; extra == "serve"
|
|
27
|
+
Requires-Dist: httpx>=0.27; extra == "serve"
|
|
28
|
+
Provides-Extra: backend
|
|
29
|
+
Requires-Dist: fastapi>=0.115; extra == "backend"
|
|
30
|
+
Requires-Dist: pydantic-settings>=2.4; extra == "backend"
|
|
31
|
+
Requires-Dist: httpx>=0.27; extra == "backend"
|
|
32
|
+
Requires-Dist: pyjwt[crypto]>=2.9; extra == "backend"
|
|
33
|
+
Requires-Dist: cryptography>=43; extra == "backend"
|
|
34
|
+
Provides-Extra: mcp
|
|
35
|
+
Requires-Dist: mcp<3,>=1.2; extra == "mcp"
|
|
36
|
+
Provides-Extra: dev
|
|
37
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
38
|
+
Requires-Dist: scipy>=1.13; extra == "dev"
|
|
39
|
+
Requires-Dist: pytest-asyncio>=0.24; extra == "dev"
|
|
40
|
+
Requires-Dist: ruff>=0.6; extra == "dev"
|
|
41
|
+
|
|
42
|
+
# Fusion First
|
|
43
|
+
|
|
44
|
+
**Measure and guard AI agents.** Fusion attacks an agent's system prompt with an OWASP-mapped suite,
|
|
45
|
+
grades safety and quality with a cross-family judge whose accuracy is measured in the same run, and
|
|
46
|
+
ships a runtime guardrail that blocks leaked secrets and tool calls the user's request does not cover
|
|
47
|
+
(measured results below). Every number carries a confidence interval or an honesty badge. The
|
|
48
|
+
prompt fix is optional and re-tested on your model (paired before/after, McNemar, honesty badge): on
|
|
49
|
+
the small open-weight models measured it rarely cut attacks and raised refusals of safe requests; see
|
|
50
|
+
the [Trust Report](https://fusion-first-testing.com/trust).
|
|
51
|
+
|
|
52
|
+
Checks (OWASP LLM Top 10 2025 / Agentic 2026): `direct_prompt_injection` (LLM01),
|
|
53
|
+
`excessive_agency` (LLM06), `data_exfiltration` (LLM02), `system_prompt_leakage` (LLM07).
|
|
54
|
+
|
|
55
|
+
## Install
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
pip install fusion-safety # core: runs, runtime guardrail, CLI, prompt fix
|
|
59
|
+
pip install "fusion-safety[mcp]" # + MCP server (fusion-mcp)
|
|
60
|
+
pip install "fusion-safety[serve]" # + local web app (fusion serve)
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
The wheel bundles the corpora (`datasets/`, `cassettes/`, `crosswalk/`, `evals/`), the web app and
|
|
64
|
+
the Claude Code plugin marketplace (`fusion plugin-dir` prints its path).
|
|
65
|
+
|
|
66
|
+
## Use
|
|
67
|
+
|
|
68
|
+
Any model or workflow can be the target:
|
|
69
|
+
|
|
70
|
+
| Target | Runs | Cost |
|
|
71
|
+
|---|---|---|
|
|
72
|
+
| `ollama:<model>` | a local model; any Hugging Face GGUF as `ollama:hf.co/<user>/<repo>` | free |
|
|
73
|
+
| `openai-compat:<url>#<model>` | your own server: vLLM, TGI, LM Studio, llama.cpp | free if local |
|
|
74
|
+
| `claude-cli:<model>` | Claude via the `claude` CLI | subscription |
|
|
75
|
+
| `openai:` `hf:` `anthropic:` `openrouter:` `together:` `groq:` `fireworks:` `mistral-api:` `deepseek:` + `<model>` | a hosted API with your key (`OPENAI_API_KEY`, `HF_TOKEN`, ...) | metered; needs `FUSION_ALLOW_API_SPEND=1` |
|
|
76
|
+
| `cmd:<command>` | any workflow: a program that reads a JSON request on stdin and prints the reply (template: [/use](https://fusion-first-testing.com/use#models)) | yours |
|
|
77
|
+
| `python:<name>` | a Python function, via `fusion_first.targets.FunctionModelClient` | yours |
|
|
78
|
+
|
|
79
|
+
Grader: `host` (the calling agent), `claude-cli`, or any chat target above. Each run measures its
|
|
80
|
+
grader on known-answer questions and withholds the grade ('?') if it falls short.
|
|
81
|
+
|
|
82
|
+
**1. Claude Code plugin** (recommended)
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
uv tool install fusion-safety # puts fusion on PATH; the plugin starts its server with uvx
|
|
86
|
+
claude plugin marketplace add "$(fusion plugin-dir)" && claude plugin install fusion@fusion-first
|
|
87
|
+
# in Claude Code: /fusion:audit prompts/support_bot.md ollama:llama3.2:1b
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
**2. Command line / CI**
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
fusion doctor # available backends
|
|
94
|
+
fusion run start --prompt agent.txt --target ollama:llama3.2:1b --grader claude-cli
|
|
95
|
+
fusion run start --prompt agent.txt --target ollama:llama3.2:1b # grader=host: pauses for grading
|
|
96
|
+
fusion run start --prompt agent.txt --target openai-compat:http://127.0.0.1:8000/v1#Qwen/Qwen2.5-7B-Instruct --grader claude-cli # vLLM / LM Studio / llama.cpp
|
|
97
|
+
fusion run start --prompt agent.txt --target hf:meta-llama/Llama-3.1-8B-Instruct --grader openai:gpt-4o # hosted, your keys
|
|
98
|
+
fusion run start --prompt agent.txt --target "cmd:python fusion_adapter.py" --grader claude-cli # any workflow
|
|
99
|
+
fusion run tasks > tasks.json; fusion run submit --file answers.json
|
|
100
|
+
fusion run finalize --min-grade B # exit 1 below the bar
|
|
101
|
+
fusion run verify # re-derive offline; exit 1 on drift
|
|
102
|
+
fusion harden --prompt agent.txt --write # optional: append the prompt fix
|
|
103
|
+
fusion run logs --file logs.jsonl --check everything --grader claude-cli # grade existing transcripts
|
|
104
|
+
fusion import-promptfoo results.json # Wilson CIs + paired McNemar on promptfoo results
|
|
105
|
+
fusion guard-bench # runtime guard benchmark (in-house corpus: 100% recall / 0% over-block)
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
No local grader is recommended: `ollama-prob:qwen2.5:7b` scored below the policy floor in its
|
|
109
|
+
pre-registered test (Trust Report).
|
|
110
|
+
|
|
111
|
+
**MCP**: `fusion-mcp` (stdio) exposes `fusion_doctor`, `start_run`, `run_status`,
|
|
112
|
+
`get_grading_tasks`, `submit_grades`, `finalize_run`, `verify_run`, the one-shot `audit_agent` /
|
|
113
|
+
`scan_prompt`, the runtime guard's `guardrail_snippet` / `check_output` / `check_tool_call`, and the
|
|
114
|
+
optional `harden_prompt`.
|
|
115
|
+
|
|
116
|
+
```json
|
|
117
|
+
{ "mcpServers": { "fusion": { "command": "fusion-mcp" } } }
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
**3. Runtime guardrail** (in your agent code; the `fusion-guard` plugin for Claude Code)
|
|
121
|
+
|
|
122
|
+
On fresh successful attacks against qwen2.5:7b and llama3.1:8b (rules 0c560b5, scored once,
|
|
123
|
+
pre-registered), the guardrail stopped 89% of real attacks while wrongly blocking 1% of clean transcripts:
|
|
124
|
+
every password leak and direct-harm hijack, 69% of data-stealing hijacks. Llama Guard 3 8B, configured,
|
|
125
|
+
stopped 52% of the same attacks. The wrapper checks replies; each tool call is checked against the
|
|
126
|
+
user's own request before it runs. Every live scan also replays the guard over its own replies
|
|
127
|
+
(in-sample): the Guard step, the HTML report and `fusion run finalize` show what it would have stopped,
|
|
128
|
+
and count separately the attacks answered in prose, where there is no tool call to check.
|
|
129
|
+
|
|
130
|
+
```python
|
|
131
|
+
from fusion_first.guardrail.guard import Guardrail
|
|
132
|
+
from fusion_first.guardrail.policy import GuardConfig
|
|
133
|
+
from fusion_first.guardrail.client import GuardedModelClient
|
|
134
|
+
|
|
135
|
+
guard = Guardrail(GuardConfig(allowlisted_domains=["your-co.com"], secret_values=["sk-..."],
|
|
136
|
+
system_prompt=SYSTEM_PROMPT, require_authorization=True)) # the measured config
|
|
137
|
+
client = GuardedModelClient(your_model_client, guard) # replies: secrets redacted, prompt dumps blocked
|
|
138
|
+
outcome = guard.guard_tool_call(tool_name, tool_args, user_request=user_message, untrusted_context=True)
|
|
139
|
+
if outcome.blocked: ... # don't run it
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
**4. Local web app**
|
|
143
|
+
|
|
144
|
+
```bash
|
|
145
|
+
pip install "fusion-safety[serve]"
|
|
146
|
+
fusion serve # http://127.0.0.1:8765; --demo-only disables live models
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
Binds 127.0.0.1, accepts only `127.0.0.1`/`localhost` Host headers, and requires a per-launch token on
|
|
150
|
+
every request that runs anything, so other websites can't drive your local models or `claude`
|
|
151
|
+
subscription.
|
|
152
|
+
|
|
153
|
+
Offline runs report `DEMONSTRATION` numbers (deterministic stand-in judge, canned responses); `--live`
|
|
154
|
+
runs through Ollama and/or the `claude` CLI. Metered API keys are used only with `--backend api` and
|
|
155
|
+
`FUSION_ALLOW_API_SPEND=1`. Full CI workflow: `integrations/README.md`.
|
|
156
|
+
|
|
157
|
+
## Develop
|
|
158
|
+
|
|
159
|
+
```bash
|
|
160
|
+
python -m pytest -q # offline suite
|
|
161
|
+
python -m ruff check fusion tests scripts
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
`fusion_first/` is the pure core (never imports `modal`); Modal/FastAPI wrappers live in `app/`. See
|
|
165
|
+
`CLAUDE.md` (project guide) and `SCHEMA.md` (data model).
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
# Fusion First
|
|
2
|
+
|
|
3
|
+
**Measure and guard AI agents.** Fusion attacks an agent's system prompt with an OWASP-mapped suite,
|
|
4
|
+
grades safety and quality with a cross-family judge whose accuracy is measured in the same run, and
|
|
5
|
+
ships a runtime guardrail that blocks leaked secrets and tool calls the user's request does not cover
|
|
6
|
+
(measured results below). Every number carries a confidence interval or an honesty badge. The
|
|
7
|
+
prompt fix is optional and re-tested on your model (paired before/after, McNemar, honesty badge): on
|
|
8
|
+
the small open-weight models measured it rarely cut attacks and raised refusals of safe requests; see
|
|
9
|
+
the [Trust Report](https://fusion-first-testing.com/trust).
|
|
10
|
+
|
|
11
|
+
Checks (OWASP LLM Top 10 2025 / Agentic 2026): `direct_prompt_injection` (LLM01),
|
|
12
|
+
`excessive_agency` (LLM06), `data_exfiltration` (LLM02), `system_prompt_leakage` (LLM07).
|
|
13
|
+
|
|
14
|
+
## Install
|
|
15
|
+
|
|
16
|
+
```bash
|
|
17
|
+
pip install fusion-safety # core: runs, runtime guardrail, CLI, prompt fix
|
|
18
|
+
pip install "fusion-safety[mcp]" # + MCP server (fusion-mcp)
|
|
19
|
+
pip install "fusion-safety[serve]" # + local web app (fusion serve)
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
The wheel bundles the corpora (`datasets/`, `cassettes/`, `crosswalk/`, `evals/`), the web app and
|
|
23
|
+
the Claude Code plugin marketplace (`fusion plugin-dir` prints its path).
|
|
24
|
+
|
|
25
|
+
## Use
|
|
26
|
+
|
|
27
|
+
Any model or workflow can be the target:
|
|
28
|
+
|
|
29
|
+
| Target | Runs | Cost |
|
|
30
|
+
|---|---|---|
|
|
31
|
+
| `ollama:<model>` | a local model; any Hugging Face GGUF as `ollama:hf.co/<user>/<repo>` | free |
|
|
32
|
+
| `openai-compat:<url>#<model>` | your own server: vLLM, TGI, LM Studio, llama.cpp | free if local |
|
|
33
|
+
| `claude-cli:<model>` | Claude via the `claude` CLI | subscription |
|
|
34
|
+
| `openai:` `hf:` `anthropic:` `openrouter:` `together:` `groq:` `fireworks:` `mistral-api:` `deepseek:` + `<model>` | a hosted API with your key (`OPENAI_API_KEY`, `HF_TOKEN`, ...) | metered; needs `FUSION_ALLOW_API_SPEND=1` |
|
|
35
|
+
| `cmd:<command>` | any workflow: a program that reads a JSON request on stdin and prints the reply (template: [/use](https://fusion-first-testing.com/use#models)) | yours |
|
|
36
|
+
| `python:<name>` | a Python function, via `fusion_first.targets.FunctionModelClient` | yours |
|
|
37
|
+
|
|
38
|
+
Grader: `host` (the calling agent), `claude-cli`, or any chat target above. Each run measures its
|
|
39
|
+
grader on known-answer questions and withholds the grade ('?') if it falls short.
|
|
40
|
+
|
|
41
|
+
**1. Claude Code plugin** (recommended)
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
uv tool install fusion-safety # puts fusion on PATH; the plugin starts its server with uvx
|
|
45
|
+
claude plugin marketplace add "$(fusion plugin-dir)" && claude plugin install fusion@fusion-first
|
|
46
|
+
# in Claude Code: /fusion:audit prompts/support_bot.md ollama:llama3.2:1b
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
**2. Command line / CI**
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
fusion doctor # available backends
|
|
53
|
+
fusion run start --prompt agent.txt --target ollama:llama3.2:1b --grader claude-cli
|
|
54
|
+
fusion run start --prompt agent.txt --target ollama:llama3.2:1b # grader=host: pauses for grading
|
|
55
|
+
fusion run start --prompt agent.txt --target openai-compat:http://127.0.0.1:8000/v1#Qwen/Qwen2.5-7B-Instruct --grader claude-cli # vLLM / LM Studio / llama.cpp
|
|
56
|
+
fusion run start --prompt agent.txt --target hf:meta-llama/Llama-3.1-8B-Instruct --grader openai:gpt-4o # hosted, your keys
|
|
57
|
+
fusion run start --prompt agent.txt --target "cmd:python fusion_adapter.py" --grader claude-cli # any workflow
|
|
58
|
+
fusion run tasks > tasks.json; fusion run submit --file answers.json
|
|
59
|
+
fusion run finalize --min-grade B # exit 1 below the bar
|
|
60
|
+
fusion run verify # re-derive offline; exit 1 on drift
|
|
61
|
+
fusion harden --prompt agent.txt --write # optional: append the prompt fix
|
|
62
|
+
fusion run logs --file logs.jsonl --check everything --grader claude-cli # grade existing transcripts
|
|
63
|
+
fusion import-promptfoo results.json # Wilson CIs + paired McNemar on promptfoo results
|
|
64
|
+
fusion guard-bench # runtime guard benchmark (in-house corpus: 100% recall / 0% over-block)
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
No local grader is recommended: `ollama-prob:qwen2.5:7b` scored below the policy floor in its
|
|
68
|
+
pre-registered test (Trust Report).
|
|
69
|
+
|
|
70
|
+
**MCP**: `fusion-mcp` (stdio) exposes `fusion_doctor`, `start_run`, `run_status`,
|
|
71
|
+
`get_grading_tasks`, `submit_grades`, `finalize_run`, `verify_run`, the one-shot `audit_agent` /
|
|
72
|
+
`scan_prompt`, the runtime guard's `guardrail_snippet` / `check_output` / `check_tool_call`, and the
|
|
73
|
+
optional `harden_prompt`.
|
|
74
|
+
|
|
75
|
+
```json
|
|
76
|
+
{ "mcpServers": { "fusion": { "command": "fusion-mcp" } } }
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
**3. Runtime guardrail** (in your agent code; the `fusion-guard` plugin for Claude Code)
|
|
80
|
+
|
|
81
|
+
On fresh successful attacks against qwen2.5:7b and llama3.1:8b (rules 0c560b5, scored once,
|
|
82
|
+
pre-registered), the guardrail stopped 89% of real attacks while wrongly blocking 1% of clean transcripts:
|
|
83
|
+
every password leak and direct-harm hijack, 69% of data-stealing hijacks. Llama Guard 3 8B, configured,
|
|
84
|
+
stopped 52% of the same attacks. The wrapper checks replies; each tool call is checked against the
|
|
85
|
+
user's own request before it runs. Every live scan also replays the guard over its own replies
|
|
86
|
+
(in-sample): the Guard step, the HTML report and `fusion run finalize` show what it would have stopped,
|
|
87
|
+
and count separately the attacks answered in prose, where there is no tool call to check.
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
from fusion_first.guardrail.guard import Guardrail
|
|
91
|
+
from fusion_first.guardrail.policy import GuardConfig
|
|
92
|
+
from fusion_first.guardrail.client import GuardedModelClient
|
|
93
|
+
|
|
94
|
+
guard = Guardrail(GuardConfig(allowlisted_domains=["your-co.com"], secret_values=["sk-..."],
|
|
95
|
+
system_prompt=SYSTEM_PROMPT, require_authorization=True)) # the measured config
|
|
96
|
+
client = GuardedModelClient(your_model_client, guard) # replies: secrets redacted, prompt dumps blocked
|
|
97
|
+
outcome = guard.guard_tool_call(tool_name, tool_args, user_request=user_message, untrusted_context=True)
|
|
98
|
+
if outcome.blocked: ... # don't run it
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
**4. Local web app**
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
pip install "fusion-safety[serve]"
|
|
105
|
+
fusion serve # http://127.0.0.1:8765; --demo-only disables live models
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Binds 127.0.0.1, accepts only `127.0.0.1`/`localhost` Host headers, and requires a per-launch token on
|
|
109
|
+
every request that runs anything, so other websites can't drive your local models or `claude`
|
|
110
|
+
subscription.
|
|
111
|
+
|
|
112
|
+
Offline runs report `DEMONSTRATION` numbers (deterministic stand-in judge, canned responses); `--live`
|
|
113
|
+
runs through Ollama and/or the `claude` CLI. Metered API keys are used only with `--backend api` and
|
|
114
|
+
`FUSION_ALLOW_API_SPEND=1`. Full CI workflow: `integrations/README.md`.
|
|
115
|
+
|
|
116
|
+
## Develop
|
|
117
|
+
|
|
118
|
+
```bash
|
|
119
|
+
python -m pytest -q # offline suite
|
|
120
|
+
python -m ruff check fusion tests scripts
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
`fusion_first/` is the pure core (never imports `modal`); Modal/FastAPI wrappers live in `app/`. See
|
|
124
|
+
`CLAUDE.md` (project guide) and `SCHEMA.md` (data model).
|