nomosguard 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
nomosguard/__init__.py ADDED
@@ -0,0 +1,9 @@
1
+ """NomosGuard: deterministic security reasoning core.
2
+
3
+ An LLM may read security events and write structured, evidence-citing
4
+ claims; this package turns those claims into tamper-evident decisions.
5
+ Language models never cross the decision boundary: the ledger, the rule
6
+ engine, and the policy gate are fully deterministic.
7
+ """
8
+
9
+ __version__ = "0.1.0"
@@ -0,0 +1,12 @@
1
+ """Attack-scenario benchmark: measure what the core catches and misses.
2
+
3
+ The corpus is deterministic and committed. run_suite.py produces a results
4
+ file from an actual run — recall, false-positive rate, and fail-closed
5
+ correctness. The honesty section in RESULTS.md states what the corpus does
6
+ not cover.
7
+ """
8
+
9
+ from .scenarios import Scenario, build_all_scenarios
10
+ from .run_suite import run_suite
11
+
12
+ __all__ = ["Scenario", "build_all_scenarios", "run_suite"]
@@ -0,0 +1,21 @@
1
+ # NomosGuard benchmark results
2
+
3
+ Run at: 2026-10-08T23:54:36.236216+00:00
4
+ Scenarios: 6
5
+
6
+ ## Metrics (measured, not asserted)
7
+
8
+ - recall: 2/2 = 100.0%
9
+ - false positives: 0
10
+ - fail-closed on incomplete evidence: OK
11
+
12
+ ## Per-scenario
13
+
14
+ | scenario | expected | derived | passed |
15
+ |---|---|---|---|
16
+ | positive_vulnerable_exposure | [['researcher', 'orders_db']] | [['researcher', 'orders_db']] | PASS |
17
+ | negative_benign_workflow | [] | [] | PASS |
18
+ | multi_agent_only_one_exposed | [['agent_a', 'orders_db']] | [['agent_a', 'orders_db']] | PASS |
19
+ | incomplete_evidence | [] | [] | PASS |
20
+ | policy_violation | [] | [] | PASS |
21
+ | no_vulnerability_no_exposure | [] | [] | PASS |
@@ -0,0 +1,81 @@
1
+ {
2
+ "scenarios": 6,
3
+ "recall": 1.0,
4
+ "caught": 2,
5
+ "expected_total": 2,
6
+ "derived_total": 2,
7
+ "false_positives": 0,
8
+ "fail_closed_ok": true,
9
+ "all_passed": true,
10
+ "run_at": "2026-10-08T23:54:36.236216+00:00",
11
+ "results": [
12
+ {
13
+ "name": "positive_vulnerable_exposure",
14
+ "expected": [
15
+ [
16
+ "researcher",
17
+ "orders_db"
18
+ ]
19
+ ],
20
+ "derived": [
21
+ [
22
+ "researcher",
23
+ "orders_db"
24
+ ]
25
+ ],
26
+ "passed": true,
27
+ "fail_closed_ok": true,
28
+ "gate_decision": "BLOCK"
29
+ },
30
+ {
31
+ "name": "negative_benign_workflow",
32
+ "expected": [],
33
+ "derived": [],
34
+ "passed": true,
35
+ "fail_closed_ok": true,
36
+ "gate_decision": "BLOCK"
37
+ },
38
+ {
39
+ "name": "multi_agent_only_one_exposed",
40
+ "expected": [
41
+ [
42
+ "agent_a",
43
+ "orders_db"
44
+ ]
45
+ ],
46
+ "derived": [
47
+ [
48
+ "agent_a",
49
+ "orders_db"
50
+ ]
51
+ ],
52
+ "passed": true,
53
+ "fail_closed_ok": true,
54
+ "gate_decision": "BLOCK"
55
+ },
56
+ {
57
+ "name": "incomplete_evidence",
58
+ "expected": [],
59
+ "derived": [],
60
+ "passed": true,
61
+ "fail_closed_ok": true,
62
+ "gate_decision": "BLOCK"
63
+ },
64
+ {
65
+ "name": "policy_violation",
66
+ "expected": [],
67
+ "derived": [],
68
+ "passed": true,
69
+ "fail_closed_ok": true,
70
+ "gate_decision": "ALERT"
71
+ },
72
+ {
73
+ "name": "no_vulnerability_no_exposure",
74
+ "expected": [],
75
+ "derived": [],
76
+ "passed": true,
77
+ "fail_closed_ok": true,
78
+ "gate_decision": "BLOCK"
79
+ }
80
+ ]
81
+ }
@@ -0,0 +1,185 @@
1
+ """Run the benchmark corpus and measure recall / false-positive / fail-closed.
2
+
3
+ Every number in the results file comes from actually running the scenarios
4
+ through the real engine and gate. Nothing is hand-computed.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ from dataclasses import dataclass, field
11
+ from datetime import datetime, timezone
12
+ from pathlib import Path
13
+ from typing import Any
14
+
15
+ from ..ledger import Claim, EvidenceLedger
16
+ from ..rules import Pattern, Rule, RuleEngine
17
+ from ..gate import Decision, PolicyGate, PolicyRule
18
+ from .scenarios import Scenario, build_all_scenarios
19
+
20
+
21
+ def default_rules() -> list[Rule]:
22
+ return [
23
+ Rule(
24
+ name="tool_on_vulnerable_component",
25
+ body=(
26
+ Pattern("?agent", "calls", "?tool"),
27
+ Pattern("?agent", "operates_on", "?component"),
28
+ Pattern("?component", "has_vulnerability", "?cve"),
29
+ ),
30
+ head=Pattern("?agent", "exposes", "?component"),
31
+ description="An agent operating a tool on a vulnerable component exposes it.",
32
+ ),
33
+ Rule(
34
+ name="policy_denied_action",
35
+ body=(
36
+ Pattern("?agent", "calls", "?tool"),
37
+ Pattern("?agent", "policy_denies", "?action"),
38
+ ),
39
+ head=Pattern("?agent", "violates", "policy"),
40
+ description="An agent calling a tool whose action is policy-denied violates policy.",
41
+ ),
42
+ ]
43
+
44
+
45
+ def default_policy_rules() -> list[PolicyRule]:
46
+ return [
47
+ PolicyRule(
48
+ name="block_vulnerable_exposure",
49
+ match_relation="exposes",
50
+ decision=Decision.BLOCK,
51
+ description="Exposure of a vulnerable component is blocked.",
52
+ ),
53
+ PolicyRule(
54
+ name="alert_policy_violation",
55
+ match_relation="violates",
56
+ decision=Decision.ALERT,
57
+ description="Policy violations raise an alert.",
58
+ ),
59
+ ]
60
+
61
+
62
+ @dataclass
63
+ class ScenarioResult:
64
+ name: str
65
+ expected: set
66
+ derived: set
67
+ passed: bool
68
+ fail_closed_ok: bool
69
+ gate_decision: str
70
+
71
+
72
+ def run_scenario(scenario: Scenario) -> ScenarioResult:
73
+ ledger = EvidenceLedger()
74
+ for spec in scenario.claims:
75
+ ledger.append(Claim(kind=spec.kind, payload=spec.payload, evidence=spec.evidence))
76
+
77
+ engine = RuleEngine(rules=default_rules())
78
+ engine.add_facts_from_ledger(ledger)
79
+ engine.derive()
80
+
81
+ derived = {
82
+ (f.subject, f.object) for f in engine.facts if f.relation == "exposes"
83
+ }
84
+ gate = PolicyGate(policy_rules=default_policy_rules(), fallback=Decision.BLOCK)
85
+ decision = gate.evaluate_with_fallback(engine)
86
+
87
+ expected = set(scenario.expected_exposed)
88
+ passed = derived == expected
89
+
90
+ # fail-closed correctness: with no derived path the gate must BLOCK
91
+ fail_closed_ok = True
92
+ if scenario.incomplete:
93
+ fail_closed_ok = (
94
+ derived == set()
95
+ and decision["decision"] == "BLOCK"
96
+ and decision["policy_rule"] == "FALLBACK"
97
+ )
98
+
99
+ return ScenarioResult(
100
+ name=scenario.name,
101
+ expected=expected,
102
+ derived=derived,
103
+ passed=passed,
104
+ fail_closed_ok=fail_closed_ok,
105
+ gate_decision=decision["decision"],
106
+ )
107
+
108
+
109
+ def run_suite(output_dir: str | Path | None = None) -> dict[str, Any]:
110
+ """Run every scenario, compute metrics, optionally write results files.
111
+
112
+ Metrics (all measured, never asserted):
113
+ - recall: caught positives / total expected positives
114
+ - false_positive: derived on scenarios expecting nothing
115
+ - fail_closed: incomplete-evidence scenarios that fell back to BLOCK
116
+ """
117
+ scenarios = build_all_scenarios()
118
+ results = [run_scenario(s) for s in scenarios]
119
+
120
+ total_expected = sum(len(r.expected) for r in results)
121
+ total_derived = sum(len(r.derived) for r in results)
122
+ caught = sum(len(r.expected & r.derived) for r in results)
123
+ false_positives = sum(
124
+ 1 for r in results if r.expected == set() and r.derived != set()
125
+ )
126
+ incomplete = [r for r in results if any(s.incomplete for s in scenarios if s.name == r.name)]
127
+ fail_closed_ok = all(r.fail_closed_ok for r in incomplete) if incomplete else True
128
+
129
+ recall = caught / total_expected if total_expected else 1.0
130
+
131
+ summary = {
132
+ "scenarios": len(results),
133
+ "recall": recall,
134
+ "caught": caught,
135
+ "expected_total": total_expected,
136
+ "derived_total": total_derived,
137
+ "false_positives": false_positives,
138
+ "fail_closed_ok": fail_closed_ok,
139
+ "all_passed": all(r.passed for r in results),
140
+ "run_at": datetime.now(timezone.utc).isoformat(),
141
+ "results": [
142
+ {
143
+ "name": r.name,
144
+ "expected": sorted(list(p) for p in r.expected),
145
+ "derived": sorted(list(p) for p in r.derived),
146
+ "passed": r.passed,
147
+ "fail_closed_ok": r.fail_closed_ok,
148
+ "gate_decision": r.gate_decision,
149
+ }
150
+ for r in results
151
+ ],
152
+ }
153
+
154
+ if output_dir is not None:
155
+ out = Path(output_dir)
156
+ out.mkdir(parents=True, exist_ok=True)
157
+ (out / "results.json").write_text(json.dumps(summary, indent=2), encoding="utf-8")
158
+ (out / "RESULTS.md").write_text(_render_markdown(summary), encoding="utf-8")
159
+
160
+ return summary
161
+
162
+
163
+ def _render_markdown(summary: dict[str, Any]) -> str:
164
+ lines = [
165
+ "# NomosGuard benchmark results",
166
+ "",
167
+ f"Run at: {summary['run_at']}",
168
+ f"Scenarios: {summary['scenarios']}",
169
+ "",
170
+ "## Metrics (measured, not asserted)",
171
+ "",
172
+ f"- recall: {summary['caught']}/{summary['expected_total']} = {summary['recall']:.1%}",
173
+ f"- false positives: {summary['false_positives']}",
174
+ f"- fail-closed on incomplete evidence: {'OK' if summary['fail_closed_ok'] else 'FAILED'}",
175
+ "",
176
+ "## Per-scenario",
177
+ "",
178
+ "| scenario | expected | derived | passed |",
179
+ "|---|---|---|---|",
180
+ ]
181
+ for r in summary["results"]:
182
+ lines.append(
183
+ f"| {r['name']} | {r['expected']} | {r['derived']} | {'PASS' if r['passed'] else 'FAIL'} |"
184
+ )
185
+ return "\n".join(lines) + "\n"
@@ -0,0 +1,37 @@
1
+ #!/usr/bin/env python3
2
+ """Entry point: python -m benchmark.run_suite_main [--output DIR]"""
3
+
4
+ import argparse
5
+
6
+ from .run_suite import run_suite
7
+
8
+
9
+ def main() -> int:
10
+ parser = argparse.ArgumentParser(description="Run the NomosGuard benchmark corpus")
11
+ parser.add_argument(
12
+ "--output",
13
+ default=None,
14
+ help="Directory to write results.json and RESULTS.md (omit to print only)",
15
+ )
16
+ args = parser.parse_args()
17
+
18
+ summary = run_suite(output_dir=args.output)
19
+ import json
20
+
21
+ print(json.dumps(
22
+ {
23
+ "scenarios": summary["scenarios"],
24
+ "recall": summary["recall"],
25
+ "caught": summary["caught"],
26
+ "expected_total": summary["expected_total"],
27
+ "false_positives": summary["false_positives"],
28
+ "fail_closed_ok": summary["fail_closed_ok"],
29
+ "all_passed": summary["all_passed"],
30
+ },
31
+ indent=2,
32
+ ))
33
+ return 0 if summary["all_passed"] else 1
34
+
35
+
36
+ if __name__ == "__main__":
37
+ raise SystemExit(main())
@@ -0,0 +1,129 @@
1
+ """The benchmark corpus: scenarios with known ground truth.
2
+
3
+ Each scenario is a small evidence set (claims the way a caller would assert
4
+ them) plus the expected outcome — the set of (agent, component) pairs the
5
+ rules SHOULD derive as "exposes", computed by hand from the facts, not by
6
+ running the engine.
7
+
8
+ Fact model (the extractor emits):
9
+ (agent, calls, tool) — the agent invoked the tool
10
+ (agent, operates_on, target) — the AGENT operates on the target
11
+
12
+ The agent-scoped operates_on is deliberate: a tool invoked by two agents
13
+ on different targets does NOT make both agents operate on both targets.
14
+ The evidence stays honest — an agent operates only on what it called.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from dataclasses import dataclass
20
+
21
+
22
+ @dataclass(frozen=True)
23
+ class ClaimSpec:
24
+ kind: str
25
+ payload: dict
26
+ evidence: str
27
+
28
+
29
+ @dataclass(frozen=True)
30
+ class Scenario:
31
+ name: str
32
+ description: str
33
+ claims: tuple[ClaimSpec, ...]
34
+ expected_exposed: frozenset # (agent, component) pairs the rules MUST derive
35
+ incomplete: bool = False # True: gate must apply its fail-closed default
36
+
37
+
38
+ def _tool_call(agent: str, tool: str, target: str, line: int) -> ClaimSpec:
39
+ return ClaimSpec(
40
+ kind="tool_call",
41
+ payload={"agent": agent, "tool": tool, "target": target},
42
+ evidence=f"log line {line}: {agent} invoked {tool} on {target}",
43
+ )
44
+
45
+
46
+ def _vuln(component: str, cve: str) -> ClaimSpec:
47
+ return ClaimSpec(
48
+ kind="vulnerability",
49
+ payload={"component": component, "cve": cve, "severity": "high"},
50
+ evidence=f"NVD record {cve} affects {component}",
51
+ )
52
+
53
+
54
+ def _policy_deny(agent: str, action: str) -> ClaimSpec:
55
+ return ClaimSpec(
56
+ kind="policy_rule",
57
+ payload={"subject": agent, "action": action, "resource": "*", "effect": "deny"},
58
+ evidence=f"policy doc: {agent} is denied {action}",
59
+ )
60
+
61
+
62
+ def build_all_scenarios() -> list[Scenario]:
63
+ """The full corpus. Deterministic: no randomness, no files, no network.
64
+
65
+ Expected sets are hand-derived from the facts:
66
+ exposed(a, c) <=> a calls some tool, a operates_on c, c has_vulnerability
67
+ """
68
+ return [
69
+ Scenario(
70
+ name="positive_vulnerable_exposure",
71
+ description="An agent calls a tool on a component with a known CVE.",
72
+ claims=(
73
+ _tool_call("researcher", "sql_query", "orders_db", 42),
74
+ _vuln("orders_db", "CVE-2026-1234"),
75
+ ),
76
+ expected_exposed=frozenset({("researcher", "orders_db")}),
77
+ ),
78
+ Scenario(
79
+ name="negative_benign_workflow",
80
+ description="A benign multi-tool workflow on non-vulnerable components must not fire.",
81
+ claims=(
82
+ _tool_call("researcher", "sql_query", "analytics_db", 1),
83
+ _tool_call("researcher", "file_reader", "/data/clean.csv", 2),
84
+ _tool_call("researcher", "chart_maker", "/data/charts", 3),
85
+ ),
86
+ expected_exposed=frozenset(),
87
+ ),
88
+ Scenario(
89
+ name="multi_agent_only_one_exposed",
90
+ description=(
91
+ "Two agents call the same tool on different targets; only one "
92
+ "target is vulnerable. The fact model must NOT make agent_b "
93
+ "exposed to agent_a's target — operates_on is agent-scoped."
94
+ ),
95
+ claims=(
96
+ _tool_call("agent_a", "sql_query", "orders_db", 1),
97
+ _tool_call("agent_b", "sql_query", "customers_db", 2),
98
+ _vuln("orders_db", "CVE-2026-1234"),
99
+ ),
100
+ expected_exposed=frozenset({("agent_a", "orders_db")}),
101
+ ),
102
+ Scenario(
103
+ name="incomplete_evidence",
104
+ description="A tool call with no vulnerability record — fail-closed default applies.",
105
+ claims=(
106
+ _tool_call("researcher", "sql_query", "orders_db", 1),
107
+ ),
108
+ expected_exposed=frozenset(),
109
+ incomplete=True,
110
+ ),
111
+ Scenario(
112
+ name="policy_violation",
113
+ description="An agent calls a tool whose action is policy-denied.",
114
+ claims=(
115
+ _tool_call("researcher", "sql_query", "orders_db", 1),
116
+ _policy_deny("researcher", "sql_query"),
117
+ ),
118
+ expected_exposed=frozenset(),
119
+ ),
120
+ Scenario(
121
+ name="no_vulnerability_no_exposure",
122
+ description="A tool call on a component with NO vulnerability record derives nothing.",
123
+ claims=(
124
+ _tool_call("researcher", "sql_query", "orders_db", 1),
125
+ _vuln("customers_db", "CVE-2026-9999"),
126
+ ),
127
+ expected_exposed=frozenset(),
128
+ ),
129
+ ]
nomosguard/demo.py ADDED
@@ -0,0 +1,88 @@
1
+ """End-to-end demo: tool-call evidence -> derivation -> fail-closed gate.
2
+
3
+ Runs a committed scenario through the full deterministic chain and prints
4
+ the real output. Run with:
5
+
6
+ python -m nomosguard.demo
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import json
12
+
13
+ from .ledger import Claim, EvidenceLedger
14
+ from .rules import Pattern, Rule, RuleEngine
15
+ from .gate import Decision, PolicyGate, PolicyRule
16
+
17
+
18
+ def build_scenario() -> EvidenceLedger:
19
+ """A committed, deterministic tool-call scenario.
20
+
21
+ Scenario: agent 'researcher' calls a data-analysis tool that operates
22
+ on a database component which has a known CVE. Policy denies access
23
+ to components with vulnerabilities.
24
+ """
25
+ ledger = EvidenceLedger()
26
+ ledger.append(
27
+ Claim(
28
+ kind="tool_call",
29
+ payload={"agent": "researcher", "tool": "sql_query", "target": "orders_db"},
30
+ evidence="agent log line 42: researcher invoked sql_query on orders_db",
31
+ )
32
+ )
33
+ ledger.append(
34
+ Claim(
35
+ kind="vulnerability",
36
+ payload={"component": "orders_db", "cve": "CVE-2026-1234", "severity": "high"},
37
+ evidence="NVD record CVE-2026-1234 affects orders_db (SQL injection)",
38
+ )
39
+ )
40
+ return ledger
41
+
42
+
43
+ def run() -> dict:
44
+ ledger = build_scenario()
45
+ ok, detail = ledger.verify()
46
+
47
+ engine = RuleEngine(
48
+ rules=[
49
+ Rule(
50
+ name="tool_on_vulnerable_component",
51
+ body=(
52
+ Pattern("?agent", "calls", "?tool"),
53
+ Pattern("?agent", "operates_on", "?component"),
54
+ Pattern("?component", "has_vulnerability", "?cve"),
55
+ ),
56
+ head=Pattern("?agent", "exposes", "?component"),
57
+ description="An agent operating a tool on a vulnerable component exposes it.",
58
+ )
59
+ ]
60
+ )
61
+ engine.add_facts_from_ledger(ledger)
62
+ derivation = engine.derive()
63
+
64
+ gate = PolicyGate(
65
+ policy_rules=[
66
+ PolicyRule(
67
+ name="block_vulnerable_exposure",
68
+ match_relation="exposes",
69
+ decision=Decision.BLOCK,
70
+ description="Exposure of a vulnerable component is blocked.",
71
+ )
72
+ ],
73
+ fallback=Decision.BLOCK,
74
+ )
75
+ result = gate.evaluate_with_fallback(engine)
76
+
77
+ return {
78
+ "ledger_verified": ok,
79
+ "ledger_detail": detail,
80
+ "entries": len(ledger.entries),
81
+ "derived_facts": [str(f) for f in engine.facts],
82
+ "derivation": derivation,
83
+ "gate_decision": result,
84
+ }
85
+
86
+
87
+ if __name__ == "__main__":
88
+ print(json.dumps(run(), indent=2))