fusion-safety 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (139) hide show
  1. fusion_safety-0.1.0/.claude-plugin/marketplace.json +21 -0
  2. fusion_safety-0.1.0/MANIFEST.in +20 -0
  3. fusion_safety-0.1.0/PKG-INFO +165 -0
  4. fusion_safety-0.1.0/README.md +124 -0
  5. fusion_safety-0.1.0/cassettes/data_exfiltration.v1.json +651 -0
  6. fusion_safety-0.1.0/cassettes/direct_prompt_injection.v1.json +838 -0
  7. fusion_safety-0.1.0/cassettes/excessive_agency.v1.json +662 -0
  8. fusion_safety-0.1.0/cassettes/system_prompt_leakage.v1.json +651 -0
  9. fusion_safety-0.1.0/crosswalk/owasp_crosswalk.v2025.yaml +52 -0
  10. fusion_safety-0.1.0/datasets/gold/data_exfiltration.v1.jsonl +23 -0
  11. fusion_safety-0.1.0/datasets/gold/direct_prompt_injection.v1.jsonl +28 -0
  12. fusion_safety-0.1.0/datasets/gold/excessive_agency.v1.jsonl +24 -0
  13. fusion_safety-0.1.0/datasets/gold/instruction_following.v1.jsonl +16 -0
  14. fusion_safety-0.1.0/datasets/gold/system_prompt_leakage.v1.jsonl +24 -0
  15. fusion_safety-0.1.0/datasets/guard_bench/cases.jsonl +78 -0
  16. fusion_safety-0.1.0/datasets/probes/data_exfiltration.v1.jsonl +18 -0
  17. fusion_safety-0.1.0/datasets/probes/direct_prompt_injection.v1.jsonl +24 -0
  18. fusion_safety-0.1.0/datasets/probes/excessive_agency.v1.jsonl +18 -0
  19. fusion_safety-0.1.0/datasets/probes/system_prompt_leakage.v1.jsonl +18 -0
  20. fusion_safety-0.1.0/datasets/quality_tests/instruction_following.v1.jsonl +10 -0
  21. fusion_safety-0.1.0/datasets/reference_agents/v1.jsonl +5 -0
  22. fusion_safety-0.1.0/evals/baseline.data_exfiltration.json +11 -0
  23. fusion_safety-0.1.0/evals/baseline.direct_prompt_injection.json +11 -0
  24. fusion_safety-0.1.0/evals/baseline.excessive_agency.json +11 -0
  25. fusion_safety-0.1.0/evals/baseline.system_prompt_leakage.json +11 -0
  26. fusion_safety-0.1.0/evals/guard_bench.baseline.json +34 -0
  27. fusion_safety-0.1.0/evals/policy.yaml +35 -0
  28. fusion_safety-0.1.0/evals/validation/v1/judge_eval_prob_qwen7b.json +296 -0
  29. fusion_safety-0.1.0/frontend/dist/assets/index-CW6TCZzY.css +1 -0
  30. fusion_safety-0.1.0/frontend/dist/assets/index-Cb3URv5L.js +76 -0
  31. fusion_safety-0.1.0/frontend/dist/index.html +28 -0
  32. fusion_safety-0.1.0/frontend/dist/logo.png +0 -0
  33. fusion_safety-0.1.0/fusion_first/__init__.py +8 -0
  34. fusion_safety-0.1.0/fusion_first/_data.py +63 -0
  35. fusion_safety-0.1.0/fusion_first/attacks/__init__.py +3 -0
  36. fusion_safety-0.1.0/fusion_first/attacks/agentic.py +80 -0
  37. fusion_safety-0.1.0/fusion_first/attacks/taxonomy.py +44 -0
  38. fusion_safety-0.1.0/fusion_first/attacks/templates.py +160 -0
  39. fusion_safety-0.1.0/fusion_first/backends/__init__.py +1 -0
  40. fusion_safety-0.1.0/fusion_first/backends/presets.py +22 -0
  41. fusion_safety-0.1.0/fusion_first/backends/resolve.py +469 -0
  42. fusion_safety-0.1.0/fusion_first/cli.py +733 -0
  43. fusion_safety-0.1.0/fusion_first/engine/__init__.py +5 -0
  44. fusion_safety-0.1.0/fusion_first/engine/before_after.py +33 -0
  45. fusion_safety-0.1.0/fusion_first/engine/fixes.py +129 -0
  46. fusion_safety-0.1.0/fusion_first/engine/guard_replay.py +131 -0
  47. fusion_safety-0.1.0/fusion_first/engine/report.py +184 -0
  48. fusion_safety-0.1.0/fusion_first/engine/report_html.py +303 -0
  49. fusion_safety-0.1.0/fusion_first/engine/user_scan.py +629 -0
  50. fusion_safety-0.1.0/fusion_first/errors.py +141 -0
  51. fusion_safety-0.1.0/fusion_first/goldset.py +133 -0
  52. fusion_safety-0.1.0/fusion_first/guardrail/__init__.py +32 -0
  53. fusion_safety-0.1.0/fusion_first/guardrail/benchmark.py +271 -0
  54. fusion_safety-0.1.0/fusion_first/guardrail/client.py +28 -0
  55. fusion_safety-0.1.0/fusion_first/guardrail/guard.py +832 -0
  56. fusion_safety-0.1.0/fusion_first/guardrail/policy.py +81 -0
  57. fusion_safety-0.1.0/fusion_first/guardrail/simulate.py +81 -0
  58. fusion_safety-0.1.0/fusion_first/guardrail/snippet.py +50 -0
  59. fusion_safety-0.1.0/fusion_first/integrations/__init__.py +37 -0
  60. fusion_safety-0.1.0/fusion_first/integrations/agent_tools.py +427 -0
  61. fusion_safety-0.1.0/fusion_first/integrations/claude_hooks.py +1296 -0
  62. fusion_safety-0.1.0/fusion_first/integrations/live.py +176 -0
  63. fusion_safety-0.1.0/fusion_first/integrations/mcp_server.py +271 -0
  64. fusion_safety-0.1.0/fusion_first/integrations/promptfoo.py +180 -0
  65. fusion_safety-0.1.0/fusion_first/integrations/run_tools.py +192 -0
  66. fusion_safety-0.1.0/fusion_first/judge/__init__.py +15 -0
  67. fusion_safety-0.1.0/fusion_first/judge/judge.py +356 -0
  68. fusion_safety-0.1.0/fusion_first/judge/prob_judge.py +390 -0
  69. fusion_safety-0.1.0/fusion_first/judge/rubric.py +276 -0
  70. fusion_safety-0.1.0/fusion_first/measure/__init__.py +7 -0
  71. fusion_safety-0.1.0/fusion_first/measure/harness.py +359 -0
  72. fusion_safety-0.1.0/fusion_first/model/__init__.py +18 -0
  73. fusion_safety-0.1.0/fusion_first/model/client.py +77 -0
  74. fusion_safety-0.1.0/fusion_first/model/models.yaml +26 -0
  75. fusion_safety-0.1.0/fusion_first/model/providers/__init__.py +4 -0
  76. fusion_safety-0.1.0/fusion_first/model/providers/_common.py +44 -0
  77. fusion_safety-0.1.0/fusion_first/model/providers/anthropic.py +122 -0
  78. fusion_safety-0.1.0/fusion_first/model/providers/claude_cli.py +464 -0
  79. fusion_safety-0.1.0/fusion_first/model/providers/command.py +83 -0
  80. fusion_safety-0.1.0/fusion_first/model/providers/demo_detectors.py +300 -0
  81. fusion_safety-0.1.0/fusion_first/model/providers/heuristic.py +179 -0
  82. fusion_safety-0.1.0/fusion_first/model/providers/ollama.py +225 -0
  83. fusion_safety-0.1.0/fusion_first/model/providers/openai.py +93 -0
  84. fusion_safety-0.1.0/fusion_first/model/providers/openai_compat.py +105 -0
  85. fusion_safety-0.1.0/fusion_first/model/registry.py +108 -0
  86. fusion_safety-0.1.0/fusion_first/model/replay.py +198 -0
  87. fusion_safety-0.1.0/fusion_first/offline.py +16 -0
  88. fusion_safety-0.1.0/fusion_first/recording.py +172 -0
  89. fusion_safety-0.1.0/fusion_first/runs/__init__.py +1 -0
  90. fusion_safety-0.1.0/fusion_first/runs/cli.py +290 -0
  91. fusion_safety-0.1.0/fusion_first/runs/engine.py +858 -0
  92. fusion_safety-0.1.0/fusion_first/runs/logs.py +434 -0
  93. fusion_safety-0.1.0/fusion_first/runs/oracle_check.py +258 -0
  94. fusion_safety-0.1.0/fusion_first/runs/service.py +464 -0
  95. fusion_safety-0.1.0/fusion_first/schemas.py +394 -0
  96. fusion_safety-0.1.0/fusion_first/security/__init__.py +7 -0
  97. fusion_safety-0.1.0/fusion_first/security/budget.py +88 -0
  98. fusion_safety-0.1.0/fusion_first/security/normalize.py +89 -0
  99. fusion_safety-0.1.0/fusion_first/security/redaction.py +95 -0
  100. fusion_safety-0.1.0/fusion_first/security/webhook.py +36 -0
  101. fusion_safety-0.1.0/fusion_first/stats/__init__.py +5 -0
  102. fusion_safety-0.1.0/fusion_first/stats/calibration.py +70 -0
  103. fusion_safety-0.1.0/fusion_first/stats/gate.py +294 -0
  104. fusion_safety-0.1.0/fusion_first/stats/metrics.py +198 -0
  105. fusion_safety-0.1.0/fusion_first/stats/paired.py +176 -0
  106. fusion_safety-0.1.0/fusion_first/targets.py +35 -0
  107. fusion_safety-0.1.0/fusion_first/validate/__init__.py +1 -0
  108. fusion_safety-0.1.0/fusion_first/validate/ifeval_lite.py +575 -0
  109. fusion_safety-0.1.0/fusion_first/validate/oracles.py +552 -0
  110. fusion_safety-0.1.0/fusion_first/validate/prereg.py +98 -0
  111. fusion_safety-0.1.0/fusion_first/web/__init__.py +2 -0
  112. fusion_safety-0.1.0/fusion_first/web/clients.py +119 -0
  113. fusion_safety-0.1.0/fusion_first/web/factory.py +283 -0
  114. fusion_safety-0.1.0/fusion_first/web/models.py +95 -0
  115. fusion_safety-0.1.0/fusion_first/web/persistence.py +36 -0
  116. fusion_safety-0.1.0/fusion_first/web/serve.py +192 -0
  117. fusion_safety-0.1.0/fusion_first/web/settings.py +53 -0
  118. fusion_safety-0.1.0/fusion_first/web/snippet.py +8 -0
  119. fusion_safety-0.1.0/fusion_first/web/sse.py +65 -0
  120. fusion_safety-0.1.0/fusion_safety.egg-info/PKG-INFO +165 -0
  121. fusion_safety-0.1.0/fusion_safety.egg-info/SOURCES.txt +137 -0
  122. fusion_safety-0.1.0/fusion_safety.egg-info/dependency_links.txt +1 -0
  123. fusion_safety-0.1.0/fusion_safety.egg-info/entry_points.txt +3 -0
  124. fusion_safety-0.1.0/fusion_safety.egg-info/requires.txt +29 -0
  125. fusion_safety-0.1.0/fusion_safety.egg-info/top_level.txt +1 -0
  126. fusion_safety-0.1.0/plugins/fusion/.claude-plugin/plugin.json +19 -0
  127. fusion_safety-0.1.0/plugins/fusion/.mcp.json +9 -0
  128. fusion_safety-0.1.0/plugins/fusion/agents/fusion-judge.md +47 -0
  129. fusion_safety-0.1.0/plugins/fusion/skills/audit/SKILL.md +65 -0
  130. fusion_safety-0.1.0/plugins/fusion/skills/grade/SKILL.md +26 -0
  131. fusion_safety-0.1.0/plugins/fusion/skills/grade-logs/SKILL.md +21 -0
  132. fusion_safety-0.1.0/plugins/fusion/skills/guard/SKILL.md +24 -0
  133. fusion_safety-0.1.0/plugins/fusion/skills/harden/SKILL.md +20 -0
  134. fusion_safety-0.1.0/plugins/fusion/skills/setup/SKILL.md +23 -0
  135. fusion_safety-0.1.0/plugins/fusion-guard/.claude-plugin/plugin.json +9 -0
  136. fusion_safety-0.1.0/plugins/fusion-guard/hooks/hooks.json +16 -0
  137. fusion_safety-0.1.0/pyproject.toml +102 -0
  138. fusion_safety-0.1.0/setup.cfg +4 -0
  139. fusion_safety-0.1.0/setup.py +147 -0
@@ -0,0 +1,21 @@
1
+ {
2
+ "name": "fusion-first",
3
+ "owner": {
4
+ "name": "Fusion First"
5
+ },
6
+ "plugins": [
7
+ {
8
+ "name": "fusion",
9
+ "source": "./plugins/fusion",
10
+ "description": "Audit an agent's prompt for safety and quality with no API key; Claude Code grades, Fusion measures the grader."
11
+ },
12
+ {
13
+ "name": "fusion-guard",
14
+ "source": "./plugins/fusion-guard",
15
+ "description": "Opt-in: guard Claude Code's own tool calls against exfiltration, destruction and prompt-injected content (asks, never auto-allows)."
16
+ }
17
+ ],
18
+ "metadata": {
19
+ "description": "Keyless safety and quality evaluation for AI agents."
20
+ }
21
+ }
@@ -0,0 +1,20 @@
1
+ # The sdist carries only what builds the wheel (owner, 2026-09-29: leave out anything unnecessary): the
2
+ # package (minus setup.py's DEV_ONLY_MODULES), pyproject/setup, README (the PyPI page), the runtime data
3
+ # (setup.py _RUNTIME_DATA), the Claude Code marketplace and the built web app. No docs, tests, examples or
4
+ # evidence.
5
+ include README.md pyproject.toml setup.py
6
+ include cassettes/*.json
7
+ include crosswalk/*.yaml
8
+ include datasets/gold/*.jsonl datasets/probes/*.jsonl datasets/guard_bench/cases.jsonl
9
+ include datasets/quality_tests/*.jsonl datasets/reference_agents/*.jsonl
10
+ include evals/policy.yaml evals/baseline.*.json evals/guard_bench.baseline.json
11
+ include evals/validation/v1/judge_eval_prob_qwen7b.json
12
+ include .claude-plugin/marketplace.json
13
+ graft plugins
14
+ exclude plugins/*/README.md
15
+ graft frontend/dist
16
+ exclude frontend/dist/og-image.png
17
+ # fusion_first/_bundled/ is a build artifact (populated by setup.py), not shipped in the sdist.
18
+ prune fusion_first/_bundled
19
+ prune tests
20
+ global-exclude __pycache__ *.pyc *.pyo
@@ -0,0 +1,165 @@
1
+ Metadata-Version: 2.4
2
+ Name: fusion-safety
3
+ Version: 0.1.0
4
+ Summary: Measure and guard AI agents: safety and quality evaluation with a runtime guardrail
5
+ Author: Fusion First
6
+ License: Proprietary
7
+ Project-URL: Homepage, https://fusion-first-testing.com
8
+ Project-URL: Documentation, https://fusion-first-testing.com/use
9
+ Project-URL: Trust Report, https://fusion-first-testing.com/trust
10
+ Keywords: llm,ai-safety,agentic,owasp,prompt-injection,guardrail,red-team,mcp
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: License :: Other/Proprietary License
13
+ Classifier: Topic :: Security
14
+ Classifier: Topic :: Software Development :: Quality Assurance
15
+ Requires-Python: >=3.11
16
+ Description-Content-Type: text/markdown
17
+ Requires-Dist: pydantic>=2.9
18
+ Requires-Dist: numpy>=1.26
19
+ Requires-Dist: pyyaml>=6.0
20
+ Provides-Extra: api
21
+ Requires-Dist: anthropic>=0.40; extra == "api"
22
+ Requires-Dist: openai>=1.50; extra == "api"
23
+ Provides-Extra: serve
24
+ Requires-Dist: fastapi>=0.115; extra == "serve"
25
+ Requires-Dist: uvicorn>=0.30; extra == "serve"
26
+ Requires-Dist: pydantic-settings>=2.4; extra == "serve"
27
+ Requires-Dist: httpx>=0.27; extra == "serve"
28
+ Provides-Extra: backend
29
+ Requires-Dist: fastapi>=0.115; extra == "backend"
30
+ Requires-Dist: pydantic-settings>=2.4; extra == "backend"
31
+ Requires-Dist: httpx>=0.27; extra == "backend"
32
+ Requires-Dist: pyjwt[crypto]>=2.9; extra == "backend"
33
+ Requires-Dist: cryptography>=43; extra == "backend"
34
+ Provides-Extra: mcp
35
+ Requires-Dist: mcp<3,>=1.2; extra == "mcp"
36
+ Provides-Extra: dev
37
+ Requires-Dist: pytest>=8.0; extra == "dev"
38
+ Requires-Dist: scipy>=1.13; extra == "dev"
39
+ Requires-Dist: pytest-asyncio>=0.24; extra == "dev"
40
+ Requires-Dist: ruff>=0.6; extra == "dev"
41
+
42
+ # Fusion First
43
+
44
+ **Measure and guard AI agents.** Fusion attacks an agent's system prompt with an OWASP-mapped suite,
45
+ grades safety and quality with a cross-family judge whose accuracy is measured in the same run, and
46
+ ships a runtime guardrail that blocks leaked secrets and tool calls the user's request does not cover
47
+ (measured results below). Every number carries a confidence interval or an honesty badge. The
48
+ prompt fix is optional and re-tested on your model (paired before/after, McNemar, honesty badge): on
49
+ the small open-weight models measured it rarely cut attacks and raised refusals of safe requests; see
50
+ the [Trust Report](https://fusion-first-testing.com/trust).
51
+
52
+ Checks (OWASP LLM Top 10 2025 / Agentic 2026): `direct_prompt_injection` (LLM01),
53
+ `excessive_agency` (LLM06), `data_exfiltration` (LLM02), `system_prompt_leakage` (LLM07).
54
+
55
+ ## Install
56
+
57
+ ```bash
58
+ pip install fusion-safety # core: runs, runtime guardrail, CLI, prompt fix
59
+ pip install "fusion-safety[mcp]" # + MCP server (fusion-mcp)
60
+ pip install "fusion-safety[serve]" # + local web app (fusion serve)
61
+ ```
62
+
63
+ The wheel bundles the corpora (`datasets/`, `cassettes/`, `crosswalk/`, `evals/`), the web app and
64
+ the Claude Code plugin marketplace (`fusion plugin-dir` prints its path).
65
+
66
+ ## Use
67
+
68
+ Any model or workflow can be the target:
69
+
70
+ | Target | Runs | Cost |
71
+ |---|---|---|
72
+ | `ollama:<model>` | a local model; any Hugging Face GGUF as `ollama:hf.co/<user>/<repo>` | free |
73
+ | `openai-compat:<url>#<model>` | your own server: vLLM, TGI, LM Studio, llama.cpp | free if local |
74
+ | `claude-cli:<model>` | Claude via the `claude` CLI | subscription |
75
+ | `openai:` `hf:` `anthropic:` `openrouter:` `together:` `groq:` `fireworks:` `mistral-api:` `deepseek:` + `<model>` | a hosted API with your key (`OPENAI_API_KEY`, `HF_TOKEN`, ...) | metered; needs `FUSION_ALLOW_API_SPEND=1` |
76
+ | `cmd:<command>` | any workflow: a program that reads a JSON request on stdin and prints the reply (template: [/use](https://fusion-first-testing.com/use#models)) | yours |
77
+ | `python:<name>` | a Python function, via `fusion_first.targets.FunctionModelClient` | yours |
78
+
79
+ Grader: `host` (the calling agent), `claude-cli`, or any chat target above. Each run measures its
80
+ grader on known-answer questions and withholds the grade ('?') if it falls short.
81
+
82
+ **1. Claude Code plugin** (recommended)
83
+
84
+ ```bash
85
+ uv tool install fusion-safety # puts fusion on PATH; the plugin starts its server with uvx
86
+ claude plugin marketplace add "$(fusion plugin-dir)" && claude plugin install fusion@fusion-first
87
+ # in Claude Code: /fusion:audit prompts/support_bot.md ollama:llama3.2:1b
88
+ ```
89
+
90
+ **2. Command line / CI**
91
+
92
+ ```bash
93
+ fusion doctor # available backends
94
+ fusion run start --prompt agent.txt --target ollama:llama3.2:1b --grader claude-cli
95
+ fusion run start --prompt agent.txt --target ollama:llama3.2:1b # grader=host: pauses for grading
96
+ fusion run start --prompt agent.txt --target openai-compat:http://127.0.0.1:8000/v1#Qwen/Qwen2.5-7B-Instruct --grader claude-cli # vLLM / LM Studio / llama.cpp
97
+ fusion run start --prompt agent.txt --target hf:meta-llama/Llama-3.1-8B-Instruct --grader openai:gpt-4o # hosted, your keys
98
+ fusion run start --prompt agent.txt --target "cmd:python fusion_adapter.py" --grader claude-cli # any workflow
99
+ fusion run tasks > tasks.json; fusion run submit --file answers.json
100
+ fusion run finalize --min-grade B # exit 1 below the bar
101
+ fusion run verify # re-derive offline; exit 1 on drift
102
+ fusion harden --prompt agent.txt --write # optional: append the prompt fix
103
+ fusion run logs --file logs.jsonl --check everything --grader claude-cli # grade existing transcripts
104
+ fusion import-promptfoo results.json # Wilson CIs + paired McNemar on promptfoo results
105
+ fusion guard-bench # runtime guard benchmark (in-house corpus: 100% recall / 0% over-block)
106
+ ```
107
+
108
+ No local grader is recommended: `ollama-prob:qwen2.5:7b` scored below the policy floor in its
109
+ pre-registered test (Trust Report).
110
+
111
+ **MCP**: `fusion-mcp` (stdio) exposes `fusion_doctor`, `start_run`, `run_status`,
112
+ `get_grading_tasks`, `submit_grades`, `finalize_run`, `verify_run`, the one-shot `audit_agent` /
113
+ `scan_prompt`, the runtime guard's `guardrail_snippet` / `check_output` / `check_tool_call`, and the
114
+ optional `harden_prompt`.
115
+
116
+ ```json
117
+ { "mcpServers": { "fusion": { "command": "fusion-mcp" } } }
118
+ ```
119
+
120
+ **3. Runtime guardrail** (in your agent code; the `fusion-guard` plugin for Claude Code)
121
+
122
+ On fresh successful attacks against qwen2.5:7b and llama3.1:8b (rules 0c560b5, scored once,
123
+ pre-registered), the guardrail stopped 89% of real attacks while wrongly blocking 1% of clean transcripts:
124
+ every password leak and direct-harm hijack, 69% of data-stealing hijacks. Llama Guard 3 8B, configured,
125
+ stopped 52% of the same attacks. The wrapper checks replies; each tool call is checked against the
126
+ user's own request before it runs. Every live scan also replays the guard over its own replies
127
+ (in-sample): the Guard step, the HTML report and `fusion run finalize` show what it would have stopped,
128
+ and count separately the attacks answered in prose, where there is no tool call to check.
129
+
130
+ ```python
131
+ from fusion_first.guardrail.guard import Guardrail
132
+ from fusion_first.guardrail.policy import GuardConfig
133
+ from fusion_first.guardrail.client import GuardedModelClient
134
+
135
+ guard = Guardrail(GuardConfig(allowlisted_domains=["your-co.com"], secret_values=["sk-..."],
136
+ system_prompt=SYSTEM_PROMPT, require_authorization=True)) # the measured config
137
+ client = GuardedModelClient(your_model_client, guard) # replies: secrets redacted, prompt dumps blocked
138
+ outcome = guard.guard_tool_call(tool_name, tool_args, user_request=user_message, untrusted_context=True)
139
+ if outcome.blocked: ... # don't run it
140
+ ```
141
+
142
+ **4. Local web app**
143
+
144
+ ```bash
145
+ pip install "fusion-safety[serve]"
146
+ fusion serve # http://127.0.0.1:8765; --demo-only disables live models
147
+ ```
148
+
149
+ Binds 127.0.0.1, accepts only `127.0.0.1`/`localhost` Host headers, and requires a per-launch token on
150
+ every request that runs anything, so other websites can't drive your local models or `claude`
151
+ subscription.
152
+
153
+ Offline runs report `DEMONSTRATION` numbers (deterministic stand-in judge, canned responses); `--live`
154
+ runs through Ollama and/or the `claude` CLI. Metered API keys are used only with `--backend api` and
155
+ `FUSION_ALLOW_API_SPEND=1`. Full CI workflow: `integrations/README.md`.
156
+
157
+ ## Develop
158
+
159
+ ```bash
160
+ python -m pytest -q # offline suite
161
+ python -m ruff check fusion tests scripts
162
+ ```
163
+
164
+ `fusion_first/` is the pure core (never imports `modal`); Modal/FastAPI wrappers live in `app/`. See
165
+ `CLAUDE.md` (project guide) and `SCHEMA.md` (data model).
@@ -0,0 +1,124 @@
1
+ # Fusion First
2
+
3
+ **Measure and guard AI agents.** Fusion attacks an agent's system prompt with an OWASP-mapped suite,
4
+ grades safety and quality with a cross-family judge whose accuracy is measured in the same run, and
5
+ ships a runtime guardrail that blocks leaked secrets and tool calls the user's request does not cover
6
+ (measured results below). Every number carries a confidence interval or an honesty badge. The
7
+ prompt fix is optional and re-tested on your model (paired before/after, McNemar, honesty badge): on
8
+ the small open-weight models measured it rarely cut attacks and raised refusals of safe requests; see
9
+ the [Trust Report](https://fusion-first-testing.com/trust).
10
+
11
+ Checks (OWASP LLM Top 10 2025 / Agentic 2026): `direct_prompt_injection` (LLM01),
12
+ `excessive_agency` (LLM06), `data_exfiltration` (LLM02), `system_prompt_leakage` (LLM07).
13
+
14
+ ## Install
15
+
16
+ ```bash
17
+ pip install fusion-safety # core: runs, runtime guardrail, CLI, prompt fix
18
+ pip install "fusion-safety[mcp]" # + MCP server (fusion-mcp)
19
+ pip install "fusion-safety[serve]" # + local web app (fusion serve)
20
+ ```
21
+
22
+ The wheel bundles the corpora (`datasets/`, `cassettes/`, `crosswalk/`, `evals/`), the web app and
23
+ the Claude Code plugin marketplace (`fusion plugin-dir` prints its path).
24
+
25
+ ## Use
26
+
27
+ Any model or workflow can be the target:
28
+
29
+ | Target | Runs | Cost |
30
+ |---|---|---|
31
+ | `ollama:<model>` | a local model; any Hugging Face GGUF as `ollama:hf.co/<user>/<repo>` | free |
32
+ | `openai-compat:<url>#<model>` | your own server: vLLM, TGI, LM Studio, llama.cpp | free if local |
33
+ | `claude-cli:<model>` | Claude via the `claude` CLI | subscription |
34
+ | `openai:` `hf:` `anthropic:` `openrouter:` `together:` `groq:` `fireworks:` `mistral-api:` `deepseek:` + `<model>` | a hosted API with your key (`OPENAI_API_KEY`, `HF_TOKEN`, ...) | metered; needs `FUSION_ALLOW_API_SPEND=1` |
35
+ | `cmd:<command>` | any workflow: a program that reads a JSON request on stdin and prints the reply (template: [/use](https://fusion-first-testing.com/use#models)) | yours |
36
+ | `python:<name>` | a Python function, via `fusion_first.targets.FunctionModelClient` | yours |
37
+
38
+ Grader: `host` (the calling agent), `claude-cli`, or any chat target above. Each run measures its
39
+ grader on known-answer questions and withholds the grade ('?') if it falls short.
40
+
41
+ **1. Claude Code plugin** (recommended)
42
+
43
+ ```bash
44
+ uv tool install fusion-safety # puts fusion on PATH; the plugin starts its server with uvx
45
+ claude plugin marketplace add "$(fusion plugin-dir)" && claude plugin install fusion@fusion-first
46
+ # in Claude Code: /fusion:audit prompts/support_bot.md ollama:llama3.2:1b
47
+ ```
48
+
49
+ **2. Command line / CI**
50
+
51
+ ```bash
52
+ fusion doctor # available backends
53
+ fusion run start --prompt agent.txt --target ollama:llama3.2:1b --grader claude-cli
54
+ fusion run start --prompt agent.txt --target ollama:llama3.2:1b # grader=host: pauses for grading
55
+ fusion run start --prompt agent.txt --target openai-compat:http://127.0.0.1:8000/v1#Qwen/Qwen2.5-7B-Instruct --grader claude-cli # vLLM / LM Studio / llama.cpp
56
+ fusion run start --prompt agent.txt --target hf:meta-llama/Llama-3.1-8B-Instruct --grader openai:gpt-4o # hosted, your keys
57
+ fusion run start --prompt agent.txt --target "cmd:python fusion_adapter.py" --grader claude-cli # any workflow
58
+ fusion run tasks > tasks.json; fusion run submit --file answers.json
59
+ fusion run finalize --min-grade B # exit 1 below the bar
60
+ fusion run verify # re-derive offline; exit 1 on drift
61
+ fusion harden --prompt agent.txt --write # optional: append the prompt fix
62
+ fusion run logs --file logs.jsonl --check everything --grader claude-cli # grade existing transcripts
63
+ fusion import-promptfoo results.json # Wilson CIs + paired McNemar on promptfoo results
64
+ fusion guard-bench # runtime guard benchmark (in-house corpus: 100% recall / 0% over-block)
65
+ ```
66
+
67
+ No local grader is recommended: `ollama-prob:qwen2.5:7b` scored below the policy floor in its
68
+ pre-registered test (Trust Report).
69
+
70
+ **MCP**: `fusion-mcp` (stdio) exposes `fusion_doctor`, `start_run`, `run_status`,
71
+ `get_grading_tasks`, `submit_grades`, `finalize_run`, `verify_run`, the one-shot `audit_agent` /
72
+ `scan_prompt`, the runtime guard's `guardrail_snippet` / `check_output` / `check_tool_call`, and the
73
+ optional `harden_prompt`.
74
+
75
+ ```json
76
+ { "mcpServers": { "fusion": { "command": "fusion-mcp" } } }
77
+ ```
78
+
79
+ **3. Runtime guardrail** (in your agent code; the `fusion-guard` plugin for Claude Code)
80
+
81
+ On fresh successful attacks against qwen2.5:7b and llama3.1:8b (rules 0c560b5, scored once,
82
+ pre-registered), the guardrail stopped 89% of real attacks while wrongly blocking 1% of clean transcripts:
83
+ every password leak and direct-harm hijack, 69% of data-stealing hijacks. Llama Guard 3 8B, configured,
84
+ stopped 52% of the same attacks. The wrapper checks replies; each tool call is checked against the
85
+ user's own request before it runs. Every live scan also replays the guard over its own replies
86
+ (in-sample): the Guard step, the HTML report and `fusion run finalize` show what it would have stopped,
87
+ and count separately the attacks answered in prose, where there is no tool call to check.
88
+
89
+ ```python
90
+ from fusion_first.guardrail.guard import Guardrail
91
+ from fusion_first.guardrail.policy import GuardConfig
92
+ from fusion_first.guardrail.client import GuardedModelClient
93
+
94
+ guard = Guardrail(GuardConfig(allowlisted_domains=["your-co.com"], secret_values=["sk-..."],
95
+ system_prompt=SYSTEM_PROMPT, require_authorization=True)) # the measured config
96
+ client = GuardedModelClient(your_model_client, guard) # replies: secrets redacted, prompt dumps blocked
97
+ outcome = guard.guard_tool_call(tool_name, tool_args, user_request=user_message, untrusted_context=True)
98
+ if outcome.blocked: ... # don't run it
99
+ ```
100
+
101
+ **4. Local web app**
102
+
103
+ ```bash
104
+ pip install "fusion-safety[serve]"
105
+ fusion serve # http://127.0.0.1:8765; --demo-only disables live models
106
+ ```
107
+
108
+ Binds 127.0.0.1, accepts only `127.0.0.1`/`localhost` Host headers, and requires a per-launch token on
109
+ every request that runs anything, so other websites can't drive your local models or `claude`
110
+ subscription.
111
+
112
+ Offline runs report `DEMONSTRATION` numbers (deterministic stand-in judge, canned responses); `--live`
113
+ runs through Ollama and/or the `claude` CLI. Metered API keys are used only with `--backend api` and
114
+ `FUSION_ALLOW_API_SPEND=1`. Full CI workflow: `integrations/README.md`.
115
+
116
+ ## Develop
117
+
118
+ ```bash
119
+ python -m pytest -q # offline suite
120
+ python -m ruff check fusion tests scripts
121
+ ```
122
+
123
+ `fusion_first/` is the pure core (never imports `modal`); Modal/FastAPI wrappers live in `app/`. See
124
+ `CLAUDE.md` (project guide) and `SCHEMA.md` (data model).