trikesh 0.6.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. trikesh-0.6.1/PKG-INFO +421 -0
  2. trikesh-0.6.1/README.md +361 -0
  3. trikesh-0.6.1/pyproject.toml +86 -0
  4. trikesh-0.6.1/setup.cfg +4 -0
  5. trikesh-0.6.1/src/trikesh/__init__.py +63 -0
  6. trikesh-0.6.1/src/trikesh/adapters/__init__.py +26 -0
  7. trikesh-0.6.1/src/trikesh/adapters/_retry.py +49 -0
  8. trikesh-0.6.1/src/trikesh/adapters/anthropic.py +58 -0
  9. trikesh-0.6.1/src/trikesh/adapters/base.py +45 -0
  10. trikesh-0.6.1/src/trikesh/adapters/compat.py +126 -0
  11. trikesh-0.6.1/src/trikesh/adapters/factory.py +134 -0
  12. trikesh-0.6.1/src/trikesh/adapters/openai.py +48 -0
  13. trikesh-0.6.1/src/trikesh/adapters/openai_compat.py +141 -0
  14. trikesh-0.6.1/src/trikesh/benchmarks/__init__.py +37 -0
  15. trikesh-0.6.1/src/trikesh/benchmarks/calibration.py +720 -0
  16. trikesh-0.6.1/src/trikesh/benchmarks/community_staging/__init__.py +82 -0
  17. trikesh-0.6.1/src/trikesh/classifiers/__init__.py +26 -0
  18. trikesh-0.6.1/src/trikesh/classifiers/base.py +84 -0
  19. trikesh-0.6.1/src/trikesh/classifiers/ensemble_classifier.py +119 -0
  20. trikesh-0.6.1/src/trikesh/classifiers/hf_classifier.py +241 -0
  21. trikesh-0.6.1/src/trikesh/classifiers/llm_classifier.py +161 -0
  22. trikesh-0.6.1/src/trikesh/classifiers/registry.py +105 -0
  23. trikesh-0.6.1/src/trikesh/classifiers/rule_classifier.py +74 -0
  24. trikesh-0.6.1/src/trikesh/cli.py +405 -0
  25. trikesh-0.6.1/src/trikesh/core/__init__.py +0 -0
  26. trikesh-0.6.1/src/trikesh/core/baseline.py +70 -0
  27. trikesh-0.6.1/src/trikesh/core/report.py +251 -0
  28. trikesh-0.6.1/src/trikesh/core/results.py +25 -0
  29. trikesh-0.6.1/src/trikesh/core/runner.py +283 -0
  30. trikesh-0.6.1/src/trikesh/core/schema.py +125 -0
  31. trikesh-0.6.1/src/trikesh/core/scorer.py +64 -0
  32. trikesh-0.6.1/src/trikesh/execution/__init__.py +19 -0
  33. trikesh-0.6.1/src/trikesh/execution/engine.py +325 -0
  34. trikesh-0.6.1/src/trikesh/execution/metrics.py +120 -0
  35. trikesh-0.6.1/src/trikesh/generators/__init__.py +0 -0
  36. trikesh-0.6.1/src/trikesh/generators/llm_generator.py +671 -0
  37. trikesh-0.6.1/src/trikesh/guard.py +504 -0
  38. trikesh-0.6.1/src/trikesh/integrations/__init__.py +67 -0
  39. trikesh-0.6.1/src/trikesh/integrations/_adk.py +69 -0
  40. trikesh-0.6.1/src/trikesh/integrations/_autogen.py +78 -0
  41. trikesh-0.6.1/src/trikesh/integrations/_base.py +47 -0
  42. trikesh-0.6.1/src/trikesh/integrations/_detector.py +50 -0
  43. trikesh-0.6.1/src/trikesh/integrations/_langchain.py +116 -0
  44. trikesh-0.6.1/src/trikesh/integrations/_wrap.py +73 -0
  45. trikesh-0.6.1/src/trikesh/judges/__init__.py +0 -0
  46. trikesh-0.6.1/src/trikesh/judges/llm_judge.py +189 -0
  47. trikesh-0.6.1/src/trikesh/judges/rule_judge.py +51 -0
  48. trikesh-0.6.1/src/trikesh/local_judge.py +410 -0
  49. trikesh-0.6.1/src/trikesh/multi_property_judge.py +437 -0
  50. trikesh-0.6.1/src/trikesh/policy/__init__.py +16 -0
  51. trikesh-0.6.1/src/trikesh/policy/dsl.py +183 -0
  52. trikesh-0.6.1/src/trikesh/probe_schemas/__init__.py +21 -0
  53. trikesh-0.6.1/src/trikesh/probe_schemas/consistency.py +105 -0
  54. trikesh-0.6.1/src/trikesh/probe_schemas/corrigibility.py +100 -0
  55. trikesh-0.6.1/src/trikesh/probe_schemas/goal_drift.py +100 -0
  56. trikesh-0.6.1/src/trikesh/probe_schemas/honesty.py +115 -0
  57. trikesh-0.6.1/src/trikesh/probe_schemas/minimal_footprint.py +84 -0
  58. trikesh-0.6.1/src/trikesh/probe_schemas/prompt_injection.py +124 -0
  59. trikesh-0.6.1/src/trikesh/probe_schemas/sycophancy.py +142 -0
  60. trikesh-0.6.1/src/trikesh/probe_schemas/trust_hierarchy.py +100 -0
  61. trikesh-0.6.1/src/trikesh.egg-info/PKG-INFO +421 -0
  62. trikesh-0.6.1/src/trikesh.egg-info/SOURCES.txt +67 -0
  63. trikesh-0.6.1/src/trikesh.egg-info/dependency_links.txt +1 -0
  64. trikesh-0.6.1/src/trikesh.egg-info/entry_points.txt +2 -0
  65. trikesh-0.6.1/src/trikesh.egg-info/requires.txt +42 -0
  66. trikesh-0.6.1/src/trikesh.egg-info/top_level.txt +1 -0
  67. trikesh-0.6.1/tests/test_adapter_factory.py +82 -0
  68. trikesh-0.6.1/tests/test_guard.py +248 -0
  69. trikesh-0.6.1/tests/test_multi_property_judge.py +116 -0
trikesh-0.6.1/PKG-INFO ADDED
@@ -0,0 +1,421 @@
1
+ Metadata-Version: 2.4
2
+ Name: trikesh
3
+ Version: 0.6.1
4
+ Summary: Behavioral regression testing and runtime safety for LLM agents
5
+ Author-email: Karan <karanxa@gmail.com>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/Karanxa/trikesh
8
+ Project-URL: Repository, https://github.com/Karanxa/trikesh
9
+ Project-URL: Bug Tracker, https://github.com/Karanxa/trikesh/issues
10
+ Keywords: llm,agent,safety,testing,behavioral,regression,sycophancy,honesty,prompt-injection,ai-safety,benchmark,evaluation,guardrails,trust
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
21
+ Classifier: Topic :: Software Development :: Testing
22
+ Classifier: Topic :: Software Development :: Quality Assurance
23
+ Requires-Python: >=3.10
24
+ Description-Content-Type: text/markdown
25
+ Requires-Dist: openai>=1.0.0
26
+ Requires-Dist: google-auth>=2.0.0
27
+ Requires-Dist: rich>=13.0.0
28
+ Requires-Dist: click>=8.1.0
29
+ Requires-Dist: pydantic>=2.0.0
30
+ Requires-Dist: numpy>=1.24.0
31
+ Requires-Dist: python-dotenv>=1.0.0
32
+ Requires-Dist: transformers>=4.40.0
33
+ Requires-Dist: huggingface_hub>=0.20.0
34
+ Requires-Dist: pyyaml>=6.0
35
+ Requires-Dist: torch>=2.1.0
36
+ Provides-Extra: dev
37
+ Requires-Dist: build; extra == "dev"
38
+ Requires-Dist: twine; extra == "dev"
39
+ Requires-Dist: pytest; extra == "dev"
40
+ Requires-Dist: pytest-asyncio; extra == "dev"
41
+ Provides-Extra: peft
42
+ Requires-Dist: peft>=0.12.0; extra == "peft"
43
+ Provides-Extra: train
44
+ Requires-Dist: torch>=2.1.0; extra == "train"
45
+ Requires-Dist: transformers>=4.45.0; extra == "train"
46
+ Requires-Dist: peft>=0.12.0; extra == "train"
47
+ Requires-Dist: datasets>=2.20.0; extra == "train"
48
+ Requires-Dist: trl>=0.11.0; extra == "train"
49
+ Requires-Dist: accelerate>=0.34.0; extra == "train"
50
+ Provides-Extra: adk
51
+ Requires-Dist: google-adk>=0.1.0; extra == "adk"
52
+ Provides-Extra: autogen
53
+ Requires-Dist: pyautogen>=0.2.0; extra == "autogen"
54
+ Provides-Extra: langchain
55
+ Requires-Dist: langchain-core>=0.1.0; extra == "langchain"
56
+ Provides-Extra: integrations
57
+ Requires-Dist: google-adk>=0.1.0; extra == "integrations"
58
+ Requires-Dist: pyautogen>=0.2.0; extra == "integrations"
59
+ Requires-Dist: langchain-core>=0.1.0; extra == "integrations"
60
+
61
+ # trikesh
62
+
63
+ **Behavioral regression testing + runtime safety for LLM agents.**
64
+
65
+ [![PyPI](https://img.shields.io/pypi/v/trikesh)](https://pypi.org/project/trikesh/)
66
+ [![License: MIT](https://img.shields.io/badge/License-MIT-green.svg)](LICENSE)
67
+ [![Python 3.10+](https://img.shields.io/badge/python-3.10+-blue.svg)](https://www.python.org/)
68
+
69
+ ---
70
+
71
+ ## What it does
72
+
73
+ trikesh solves two distinct problems:
74
+
75
+ **1. Behavioral regression testing** — measure model behavior across 8 properties (sycophancy, honesty, consistency, prompt injection, trust hierarchy, minimal footprint, goal drift, corrigibility). Run structured probes and get coverage reports.
76
+
77
+ **2. Runtime agent safety** — intercept unsafe tool calls before execution. Compose a safety stack (rules → classifiers → LLMs). Plug in any judge: local models, HuggingFace, API-based LLMs, custom matchers, ensembles.
78
+
79
+ ---
80
+
81
+ ## The 8 behavioral properties
82
+
83
+ These are what trikesh measures — independently of which classifier you use:
84
+
85
+ | Property | What it checks |
86
+ |---|---|
87
+ | **Sycophancy** | Does the model abandon correct answers when users push back? |
88
+ | **Honesty** | Does the model acknowledge uncertainty instead of confabulating? |
89
+ | **Consistency** | Do equivalent questions get equivalent answers? |
90
+ | **Prompt Injection** | Does the model follow instructions embedded in external data? |
91
+ | **Trust Hierarchy** | Does the model respect operator rules over user requests? |
92
+ | **Minimal Footprint** | Does the model prefer reversible actions over irreversible ones? |
93
+ | **Goal Drift** | Does the model stay on task or expand scope without permission? |
94
+ | **Corrigibility** | Does the model stop when told to stop? |
95
+
96
+ ---
97
+
98
+ ## Why this matters
99
+
100
+ Behavioral safety isn't a jailbreak problem — it's a *values* problem.
101
+
102
+ Sycophancy, goal drift, prompt injection, corrigibility failures don't show up in accuracy benchmarks. They show up when users push back, when prompts change, when you swap providers. And by then, it's in production.
103
+
104
+ The [MASK Benchmark (2026)](https://arxiv.org/abs/2503.03750) found:
105
+ - No frontier model is honest **more than 46% of the time** under social pressure
106
+ - Larger models are *less* honest, not more
107
+ - **83% of models** self-report knowing they contradicted their own beliefs
108
+
109
+ trikesh measures this. Before it reaches your users.
110
+
111
+ ---
112
+
113
+ ## Installation
114
+
115
+ ```bash
116
+ pip install trikesh
117
+ ```
118
+
119
+ ---
120
+
121
+ ## Benchmarking
122
+
123
+ ```bash
124
+ # Run the static bench-v1 benchmark (reproducible, citable)
125
+ trikesh run --model gpt-4o-mini --benchmark bench-v1
126
+
127
+ # Generate dynamic probes
128
+ trikesh run --model gpt-4o-mini
129
+
130
+ # Compare two models side by side
131
+ trikesh compare --models gpt-4o-mini,claude-3-5-sonnet-20241022
132
+
133
+ # Check your judge's accuracy against ground truth
134
+ trikesh calibrate --judge-model gpt-4o-mini
135
+ ```
136
+
137
+ ### Benchmarking with trikesh
138
+
139
+ trikesh includes **bench-v1**, a static set of 96 hand-authored probes grounded in safety research. Use it to evaluate any model:
140
+
141
+ ```python
142
+ from trikesh.benchmarks import load_benchmark
143
+
144
+ bench = load_benchmark("bench-v1")
145
+ # {"version": "bench-v1", "count": 96, "properties": [...]}
146
+ ```
147
+
148
+ Results are reproducible and comparable across teams — useful as a reference when comparing models or evaluating your own classifiers.
149
+
150
+ ---
151
+
152
+ ## Architecture
153
+
154
+ trikesh v0.5+ uses a **pluggable, policy-driven architecture**:
155
+
156
+ - **Classifiers**: Pluggable safety judges. Plug in LLM-based judges, rule-based matchers, HuggingFace models, or custom classifiers via a simple interface.
157
+ - **Policy DSL**: Declarative YAML policies define which classifiers run at which execution layers, with confidence thresholds and fallback chains.
158
+ - **ExecutionEngine**: Orchestrates classifiers across properties with two strategies:
159
+ - **Cascade**: Try each layer's classifiers in order; stop at first confident result
160
+ - **Speculative**: Run concurrent classifiers in a layer; use first confident winner (lower latency)
161
+ - **Observable**: Every classifier invocation is tracked — latency, confidence, outcome — accessible via `guard.metrics`
162
+
163
+ **Backwards compatible**: The legacy `SafetyGuard(mode=..., judge_model=...)` API still works unchanged.
164
+
165
+ ### PDP / PEP model
166
+
167
+ trikesh's runtime guard follows the same decision/enforcement split used in access-control systems (XACML, OPA) and increasingly in agent-authorization tools (AWS Cedar, NVIDIA OpenShell):
168
+
169
+ - **`SafetyGuard` is the Policy Decision Point (PDP).** It takes a proposed action + context and returns a structured verdict (`SafetyCheckResult`) — it never touches the live system itself. The decision logic is pluggable: any LLM via `ModelAdapter`, or any `Classifier` (rule-based, HuggingFace, local, ensemble). This makes trikesh a *behavioral-judgment* PDP — it reasons about pressure, honesty, consistency, and goal drift rather than matching static rules — distinct from and complementary to deterministic-permission PDPs (allowlists, spending ceilings) you might run alongside it.
170
+ - **`wrap()` / `protect()` are the Policy Enforcement Points (PEP).** They intercept the actual tool call — outside the model's own reasoning, so a compromised or manipulated agent can't talk its way past the check — query the PDP, and enforce its verdict (raise `SafetyBlockedError` or let the call through). These are framework-*aware* wrappers, not code merged into LangChain/AutoGen/ADK themselves: each adapter (e.g. `LangChainAdapter`) targets that framework's real tool-invocation point from outside, the same way most production PEPs work (a gateway, a proxy, an OS boundary — not literally inside the thing they're protecting).
171
+ - **The Policy DSL (SPML) is the Policy Administration Point (PAP).** Policies are authored as plain, flat YAML — deliberately with no expressions or embedded logic — decoupled from both the decision engine and the enforcement layer.
172
+
173
+ There's no formal Policy Information Point (PIP) yet — context, operator constraints, and goals are passed as explicit parameters today rather than gathered through a pluggable attribute provider. That's a known gap, not a shipped feature.
174
+
175
+ ---
176
+
177
+ ## Runtime SafetyGuard
178
+
179
+ ### Legacy API (still works)
180
+
181
+ Add one check before your agent executes any action:
182
+
183
+ ```python
184
+ from trikesh import SafetyGuard
185
+
186
+ guard = SafetyGuard()
187
+
188
+ result = guard.check(
189
+ action="DELETE FROM users WHERE last_login < '2023-01-01'",
190
+ context="Production database agent",
191
+ operator_constraints=[
192
+ "Never DELETE on production without explicit written confirmation",
193
+ ],
194
+ )
195
+
196
+ if not result.is_safe:
197
+ # Don't execute — show violations to the user or log them
198
+ for v in result.violations:
199
+ print(f"[{v.severity.upper()}] {v.description}")
200
+ ```
201
+
202
+ ```python
203
+ # Async pipelines
204
+ result = await guard.acheck(action="...", context="...")
205
+ ```
206
+
207
+ ### Policy-Driven API (new)
208
+
209
+ Use declarative policies for fine-grained control:
210
+
211
+ ```python
212
+ from trikesh import SafetyGuard, Policy
213
+
214
+ # Load a pre-built policy
215
+ policy = Policy.from_yaml("policies/default.yml")
216
+ guard = SafetyGuard(policy=policy)
217
+
218
+ # Or define one in code
219
+ from trikesh.policy import Policy, PolicyProperty, ExecutionLayer
220
+
221
+ policy = Policy(
222
+ version="1.0",
223
+ policy_id="my-policy",
224
+ properties=[
225
+ PolicyProperty(
226
+ name="sycophancy",
227
+ classifier="llm:gpt-4o-mini",
228
+ fallback="rule:capitulation",
229
+ ),
230
+ ],
231
+ execution={
232
+ "balanced": [
233
+ ExecutionLayer(
234
+ name="fast",
235
+ classifiers=["rule:basic_checks"],
236
+ timeout_ms=10,
237
+ strategy="cascade",
238
+ ),
239
+ ExecutionLayer(
240
+ name="thorough",
241
+ classifiers=["llm:gpt-4o-mini"],
242
+ timeout_ms=2000,
243
+ strategy="cascade",
244
+ ),
245
+ ]
246
+ },
247
+ )
248
+
249
+ guard = SafetyGuard(policy=policy)
250
+ result = await guard.acheck(action="...", context="...", mode="balanced")
251
+
252
+ # Inspect which classifiers were used
253
+ print(guard.metrics.summary())
254
+ ```
255
+
256
+ ### Pluggable Classifiers
257
+
258
+ trikesh ships with built-in classifiers and supports custom ones:
259
+
260
+ ```python
261
+ from trikesh.classifiers import ClassifierRegistry, HFModelClassifier
262
+
263
+ # Use HuggingFace models
264
+ hf_classifier = HFModelClassifier("Qwen/Qwen2.5-0.5B")
265
+ ClassifierRegistry.register("hf:qwen-0.5b", hf_classifier)
266
+
267
+ # Use saroku-guard, the local PDP model — resolved by its built-in id
268
+ local = ClassifierRegistry.resolve("local:saroku-safety")
269
+
270
+ # Combine classifiers in an ensemble — register custom instances under "custom:"
271
+ from trikesh.classifiers import EnsembleClassifier
272
+ ensemble = EnsembleClassifier(
273
+ classifiers=[local, hf_classifier],
274
+ strategy="majority", # or "cascade"
275
+ )
276
+ ClassifierRegistry.register("custom:hybrid", ensemble)
277
+ ```
278
+
279
+ ### Modes
280
+
281
+ ```python
282
+ # Default — saroku-guard (local PDP) protects every call immediately, no
283
+ # setup required. Safe actions are cleared in ~10-30ms with no API cost.
284
+ # Anything flagged escalates to the LLM judge, if a provider key is set,
285
+ # for full attribution across all 8 behavioral properties.
286
+ guard = SafetyGuard()
287
+
288
+ # Local PDP only — zero API calls, works fully offline.
289
+ guard = SafetyGuard(mode="local")
290
+
291
+ # Skip the local PDP and always use the full LLM judge.
292
+ guard = SafetyGuard(mode="thorough", judge_model="gpt-4o-mini")
293
+
294
+ # Tune the block threshold — lower it to block more aggressively,
295
+ # raise it to reduce false-positive blocks.
296
+ guard = SafetyGuard(local_threshold=0.3)
297
+ ```
298
+
299
+ ### What gets blocked
300
+
301
+ ```
302
+ ⛔ BLOCKED DELETE inactive users from production (no dry-run, no confirmation)
303
+ ⛔ BLOCKED Deploy with skip_tests=True (safety bypass)
304
+ ⛔ BLOCKED Grant admin — no approval ticket (constraint violation)
305
+ ⛔ BLOCKED Disable rate limiting (irreversible risk)
306
+ ✅ ALLOWED SELECT COUNT(*) — read-only query
307
+ ✅ ALLOWED Grant read access — ticket: JIRA-5821
308
+ ✅ ALLOWED Read service config
309
+ ```
310
+
311
+ ### Framework Integration
312
+
313
+ trikesh integrates with popular agent frameworks — wrap tools or entire agents:
314
+
315
+ ```python
316
+ from trikesh import wrap, protect
317
+
318
+ # Protect a single tool
319
+ safe_search = wrap(agent.search_tool, guard=guard)
320
+
321
+ # Protect all tools in an agent (auto-detects framework)
322
+ from trikesh import SafetyBlockedError
323
+ safe_agent = await protect(agent, guard=guard)
324
+
325
+ # Handle blocked actions
326
+ try:
327
+ result = await safe_agent.run(task)
328
+ except SafetyBlockedError as e:
329
+ print(f"Action blocked: {e.violations}")
330
+ ```
331
+
332
+ Supported frameworks: **Google ADK, AutoGen, LangChain**
333
+
334
+ ### Observability
335
+
336
+ Every classifier invocation is tracked automatically:
337
+
338
+ ```python
339
+ # After running checks
340
+ metrics = guard.metrics
341
+
342
+ # Get a summary
343
+ print(metrics.summary())
344
+ # {
345
+ # "total_invocations": 42,
346
+ # "by_classifier": {"llm:gpt-4o-mini": 23, "rule:basic": 19},
347
+ # "avg_latency_ms": 145.2,
348
+ # "confident_rate": 0.88,
349
+ # "timeout_rate": 0.02,
350
+ # }
351
+
352
+ # Get raw invocations for detailed analysis
353
+ for invocation in metrics.to_list():
354
+ print(f"{invocation.classifier_id}: {invocation.latency_ms}ms, confidence={invocation.confidence}")
355
+ ```
356
+
357
+ ### Performance
358
+
359
+ | Scenario | Latency |
360
+ |---|---|
361
+ | Clear violation caught by rules | <1ms |
362
+ | Action evaluated by local PDP model | ~10-30ms |
363
+ | Avg across 1000 queries (cascade) | <50ms |
364
+ | Speculative layer (concurrent) | ~max(fastest, all_uncertain) |
365
+
366
+ ---
367
+
368
+ ## Local PDP model — saroku-guard
369
+
370
+ [saroku-guard](https://huggingface.co/karanxa/saroku-guard) protects every `SafetyGuard()` by default — no setup, no API key, no data leaving your environment. It downloads automatically on first use and runs on CPU.
371
+
372
+ ```python
373
+ guard = SafetyGuard() # saroku-guard is already active
374
+ ```
375
+
376
+ To use a different checkpoint, or disable it in favor of an LLM-only judge:
377
+
378
+ ```python
379
+ guard = SafetyGuard(local_model_path="your-org/your-model")
380
+ guard = SafetyGuard(use_local_pdp=False, judge_model="gpt-4o-mini")
381
+ ```
382
+
383
+ ### Train your own
384
+
385
+ If you want to fine-tune on your own data or domain:
386
+
387
+ ```bash
388
+ pip install trikesh[train]
389
+ python -m trikesh.training.trainer --output-dir ./my-model --epochs 3
390
+ ```
391
+
392
+ ---
393
+
394
+ ## Result object
395
+
396
+ ```python
397
+ result = guard.check(...)
398
+
399
+ result.is_safe # bool
400
+ result.violations # list of SafetyViolation
401
+ result.latency_ms # float
402
+ result.layers_used # ["rules", "ml", "local_model"]
403
+ result.ml_risk_score # float 0-1
404
+ result.summary() # human-readable string
405
+ ```
406
+
407
+ Each `SafetyViolation`:
408
+
409
+ ```python
410
+ v.property # "trust_hierarchy", "minimal_footprint", etc.
411
+ v.severity # "high", "medium", "low"
412
+ v.description # what the violation is
413
+ v.recommendation # what to do instead
414
+ v.source # "rules", "ml", or "local_model"
415
+ ```
416
+
417
+ ---
418
+
419
+ ## License
420
+
421
+ MIT