@heretek-ai/epistemic-swarm 0.2.2 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/skills/brainstorming/SKILL.md +13 -0
- package/.agents/skills/code_audit/SKILL.md +13 -0
- package/.agents/skills/epistemic_search/SKILL.md +13 -0
- package/.agents/skills/grilling/SKILL.md +13 -0
- package/.agents/skills/oss_scout/SKILL.md +13 -0
- package/.agents/skills/research_cache/SKILL.md +13 -0
- package/.agents/skills/swarm_config/SKILL.md +13 -0
- package/.claude-plugin/plugin.json +15 -5
- package/.omp/README.md +39 -0
- package/.omp/SYSTEM.md +12 -0
- package/.omp/commands/audit.md +10 -0
- package/.omp/commands/brainstorming.md +12 -0
- package/.omp/commands/grill.md +9 -0
- package/.omp/commands/scout.md +10 -0
- package/.omp/commands/swarm-config.md +10 -0
- package/.omp/commands/swarm.md +9 -0
- package/.omp/hooks/post/epistemic-audit.ts +16 -0
- package/.omp/hooks/pre/epistemic-redirect.ts +21 -0
- package/.omp/prompts/brainstorming.md +8 -0
- package/.omp/prompts/swarm.md +8 -0
- package/MARKETPLACE.md +8 -0
- package/README.md +53 -8
- package/bin/cli.js +27 -3
- package/config/domain_packs/biopharma.json +23 -0
- package/config/domain_packs/legal.json +19 -0
- package/config/domain_packs/quant.json +19 -0
- package/config/mcp-research-servers.json +7 -0
- package/config/mcp_launcher.py +48 -137
- package/config/opencode-snippet.json +58 -3
- package/config/searxng_mcp.py +42 -83
- package/extensions/pi/index.js +196 -28
- package/install.sh +20 -4
- package/package.json +40 -5
- package/plugins/antigravity/README.md +28 -0
- package/plugins/antigravity/agents/alpha-thesis.md +6 -0
- package/plugins/antigravity/agents/beta-antithesis.md +7 -0
- package/plugins/antigravity/agents/brainstormer.md +7 -0
- package/plugins/antigravity/agents/epistemic-auditor.md +5 -0
- package/plugins/antigravity/hooks.json +23 -0
- package/plugins/antigravity/mcp_config.json +33 -0
- package/plugins/antigravity/plugin.json +21 -0
- package/plugins/antigravity/rules/epistemic-integrity.md +6 -0
- package/plugins/antigravity/skills/brainstorming/SKILL.md +13 -0
- package/plugins/antigravity/skills/code_audit/SKILL.md +13 -0
- package/plugins/antigravity/skills/epistemic_search/SKILL.md +13 -0
- package/plugins/antigravity/skills/grilling/SKILL.md +13 -0
- package/plugins/antigravity/skills/oss_scout/SKILL.md +13 -0
- package/plugins/antigravity/skills/research_cache/SKILL.md +13 -0
- package/plugins/antigravity/skills/swarm_config/SKILL.md +13 -0
- package/plugins/codex/AGENTS.md.snippet +10 -0
- package/plugins/codex/README.md +37 -0
- package/plugins/codex/config.toml.snippet +28 -0
- package/plugins/codex/openai.yaml +24 -0
- package/plugins/codex/skills/brainstorming/SKILL.md +13 -0
- package/plugins/codex/skills/code_audit/SKILL.md +13 -0
- package/plugins/codex/skills/epistemic_search/SKILL.md +13 -0
- package/plugins/codex/skills/grilling/SKILL.md +13 -0
- package/plugins/codex/skills/oss_scout/SKILL.md +13 -0
- package/plugins/codex/skills/research_cache/SKILL.md +13 -0
- package/plugins/codex/skills/swarm_config/SKILL.md +13 -0
- package/plugins/gemini/GEMINI.md +15 -0
- package/plugins/gemini/README.md +19 -0
- package/plugins/gemini/commands/audit.toml +6 -0
- package/plugins/gemini/commands/brainstorming.toml +10 -0
- package/plugins/gemini/commands/grill.toml +6 -0
- package/plugins/gemini/commands/scout.toml +7 -0
- package/plugins/gemini/commands/swarm-config.toml +7 -0
- package/plugins/gemini/commands/swarm.toml +8 -0
- package/plugins/gemini/gemini-extension.json +38 -0
- package/plugins/gemini/hooks/hooks.json +11 -0
- package/plugins/gemini/skills/brainstorming/SKILL.md +13 -0
- package/plugins/gemini/skills/code_audit/SKILL.md +13 -0
- package/plugins/gemini/skills/epistemic_search/SKILL.md +13 -0
- package/plugins/gemini/skills/grilling/SKILL.md +13 -0
- package/plugins/gemini/skills/oss_scout/SKILL.md +13 -0
- package/plugins/gemini/skills/research_cache/SKILL.md +13 -0
- package/plugins/gemini/skills/swarm_config/SKILL.md +13 -0
- package/plugins/opencode/index.js +335 -118
- package/prompts/agent_brainstormer.md +97 -0
- package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
- package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
- package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
- package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
- package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
- package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
- package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
- package/runner/auctioneer.py +169 -0
- package/runner/auditor_engine.py +127 -12
- package/runner/claim_store.py +361 -0
- package/runner/claim_witness.py +183 -0
- package/runner/living_dossiers.py +355 -0
- package/runner/mcp_protocol.py +188 -0
- package/runner/mcp_server.py +567 -0
- package/runner/pcrb.py +212 -0
- package/runner/pcrb_verify.py +237 -0
- package/runner/refinement.py +335 -0
- package/runner/research_swarm.py +378 -94
- package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
- package/runner/tests/fixtures/auction_objective.json +35 -0
- package/runner/tests/fixtures/borderline_claims.json +27 -0
- package/runner/tests/fixtures/divergence_objectives.json +31 -0
- package/runner/tests/test_auction_order.py +173 -0
- package/runner/tests/test_claim_store.py +162 -0
- package/runner/tests/test_claim_witness.py +186 -0
- package/runner/tests/test_domain_packs.py +217 -0
- package/runner/tests/test_fleet_seam.py +113 -0
- package/runner/tests/test_living_dossiers.py +204 -0
- package/runner/tests/test_mcp_server.py +212 -0
- package/runner/tests/test_pcrb.py +240 -0
- package/runner/tests/test_refinement.py +255 -0
- package/runner/tests/test_swarm.py +173 -15
- package/scripts/auction_experiment.py +180 -0
- package/scripts/build_adapters.py +183 -0
- package/scripts/divergence_experiment.py +184 -0
- package/skills/brainstorming/SKILL.md +106 -0
- package/skills/brainstorming/__init__.py +1 -0
- package/skills/brainstorming/scripts/brainstorm.py +200 -0
- package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
- package/skills/swarm_config/SKILL.md +1 -1
- package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
- package/skills/swarm_config/configure.py +45 -11
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Tests for the first-party MCP server (runner/mcp_server.py) and the shared
|
|
3
|
+
protocol loop (runner/mcp_protocol.py)."""
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import subprocess
|
|
7
|
+
import sys
|
|
8
|
+
import tempfile
|
|
9
|
+
import unittest
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
|
|
13
|
+
if str(PROJECT_ROOT) not in sys.path:
|
|
14
|
+
sys.path.insert(0, str(PROJECT_ROOT))
|
|
15
|
+
|
|
16
|
+
from runner.mcp_protocol import StdioJsonRpcServer, ToolSpec, read_message, write_message # noqa: E402
|
|
17
|
+
from runner.mcp_server import SERVER_NAME, build_server, build_tools # noqa: E402
|
|
18
|
+
|
|
19
|
+
EXPECTED_TOOLS = [
|
|
20
|
+
"iumbtems_config",
|
|
21
|
+
"iumbtems_swarm_research",
|
|
22
|
+
"iumbtems_code_audit",
|
|
23
|
+
"iumbtems_oss_scout",
|
|
24
|
+
"iumbtems_brainstorm",
|
|
25
|
+
"iumbtems_verify_quote",
|
|
26
|
+
"iumbtems_socratic_frontier",
|
|
27
|
+
"iumbtems_reindex_claims",
|
|
28
|
+
"iumbtems_report_retraction",
|
|
29
|
+
"iumbtems_check_staleness",
|
|
30
|
+
"iumbtems_set_domain_pack",
|
|
31
|
+
"iumbtems_export_brief",
|
|
32
|
+
"iumbtems_verify_brief",
|
|
33
|
+
]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class TestMcpProtocol(unittest.TestCase):
|
|
37
|
+
def test_tool_spec_manifest_shape(self):
|
|
38
|
+
spec = ToolSpec("t", "d", {"type": "object"}, lambda a: "x")
|
|
39
|
+
self.assertEqual(spec.to_manifest(), {"name": "t", "description": "d", "inputSchema": {"type": "object"}})
|
|
40
|
+
|
|
41
|
+
def test_alias_dispatch(self):
|
|
42
|
+
calls = []
|
|
43
|
+
spec = ToolSpec("main", "d", {}, lambda a: calls.append(a) or "ok", aliases=("alt",))
|
|
44
|
+
server = StdioJsonRpcServer("s", "1", [spec])
|
|
45
|
+
res = server.handle({"jsonrpc": "2.0", "id": 1, "method": "tools/call",
|
|
46
|
+
"params": {"name": "alt", "arguments": {"q": 1}}})
|
|
47
|
+
self.assertEqual(res["result"]["content"][0]["text"], "ok")
|
|
48
|
+
self.assertEqual(calls, [{"q": 1}])
|
|
49
|
+
|
|
50
|
+
def test_handler_exception_becomes_error_not_crash(self):
|
|
51
|
+
def boom(_):
|
|
52
|
+
raise RuntimeError("kaboom")
|
|
53
|
+
|
|
54
|
+
server = StdioJsonRpcServer("s", "1", [ToolSpec("t", "d", {}, boom)])
|
|
55
|
+
res = server.handle({"jsonrpc": "2.0", "id": 2, "method": "tools/call",
|
|
56
|
+
"params": {"name": "t", "arguments": {}}})
|
|
57
|
+
self.assertIn("error", res)
|
|
58
|
+
self.assertEqual(res["error"]["code"], -32603)
|
|
59
|
+
self.assertIn("kaboom", res["error"]["message"])
|
|
60
|
+
|
|
61
|
+
def test_notification_gets_no_reply(self):
|
|
62
|
+
server = StdioJsonRpcServer("s", "1", [])
|
|
63
|
+
self.assertIsNone(server.handle({"jsonrpc": "2.0", "method": "notifications/initialized"}))
|
|
64
|
+
|
|
65
|
+
def test_unknown_method_is_32601(self):
|
|
66
|
+
server = StdioJsonRpcServer("s", "1", [])
|
|
67
|
+
res = server.handle({"jsonrpc": "2.0", "id": 3, "method": "bogus/x"})
|
|
68
|
+
self.assertEqual(res["error"]["code"], -32601)
|
|
69
|
+
|
|
70
|
+
def test_content_length_round_trip(self):
|
|
71
|
+
import io
|
|
72
|
+
|
|
73
|
+
body = json.dumps({"jsonrpc": "2.0", "id": 1, "method": "initialize", "params": {}})
|
|
74
|
+
raw = f"Content-Length: {len(body)}\r\n\r\n{body}"
|
|
75
|
+
req, use_headers = read_message(io.StringIO(raw))
|
|
76
|
+
self.assertTrue(use_headers)
|
|
77
|
+
self.assertEqual(req["method"], "initialize")
|
|
78
|
+
|
|
79
|
+
out = io.StringIO()
|
|
80
|
+
write_message(out, {"jsonrpc": "2.0", "id": 1, "result": {}}, use_headers=True)
|
|
81
|
+
self.assertTrue(out.getvalue().startswith("Content-Length:"))
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
class TestMcpServerTools(unittest.TestCase):
|
|
85
|
+
def test_canonical_tool_names(self):
|
|
86
|
+
names = [t.name for t in build_tools()]
|
|
87
|
+
self.assertEqual(sorted(names), sorted(EXPECTED_TOOLS))
|
|
88
|
+
|
|
89
|
+
def test_server_meta(self):
|
|
90
|
+
server = build_server()
|
|
91
|
+
self.assertEqual(server.server_name, SERVER_NAME)
|
|
92
|
+
self.assertEqual(server.protocol_version, "2024-11-05")
|
|
93
|
+
self.assertEqual(sorted(t.name for t in server.tools), sorted(EXPECTED_TOOLS))
|
|
94
|
+
|
|
95
|
+
def test_tools_list_via_stdio_handshake(self):
|
|
96
|
+
"""Subprocess handshake: initialize -> tools/list -> tools/call."""
|
|
97
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
98
|
+
tmp_path = Path(tmp)
|
|
99
|
+
# Seed a source so verify_quote has something real to check.
|
|
100
|
+
sys.path.insert(0, str(PROJECT_ROOT))
|
|
101
|
+
from skills.research_cache.hasher import SourceHasher
|
|
102
|
+
|
|
103
|
+
hasher = SourceHasher(tmp_path)
|
|
104
|
+
content = "The FPGA pipeline executes Poseidon in 184ms."
|
|
105
|
+
digest = hasher.store_source(url="https://example.test/p", content=content, title="T")
|
|
106
|
+
|
|
107
|
+
# Drive handle() directly in-process for tools/call; use the
|
|
108
|
+
# subprocess only to prove the stdio loop serves initialize/tools/list.
|
|
109
|
+
proc = subprocess.Popen(
|
|
110
|
+
[sys.executable, str(PROJECT_ROOT / "runner" / "mcp_server.py")],
|
|
111
|
+
stdin=subprocess.PIPE,
|
|
112
|
+
stdout=subprocess.PIPE,
|
|
113
|
+
stderr=subprocess.PIPE,
|
|
114
|
+
text=True,
|
|
115
|
+
)
|
|
116
|
+
lines = [
|
|
117
|
+
json.dumps({"jsonrpc": "2.0", "id": 1, "method": "initialize", "params": {}}),
|
|
118
|
+
json.dumps({"jsonrpc": "2.0", "id": 2, "method": "tools/list", "params": {}}),
|
|
119
|
+
]
|
|
120
|
+
out, err = proc.communicate("\n".join(lines) + "\n", timeout=20)
|
|
121
|
+
self.assertEqual(err.strip(), "", f"server stderr: {err}")
|
|
122
|
+
responses = [json.loads(l) for l in out.strip().splitlines() if l.strip()]
|
|
123
|
+
self.assertEqual(len(responses), 2)
|
|
124
|
+
init, tools_list = responses
|
|
125
|
+
self.assertEqual(init["result"]["protocolVersion"], "2024-11-05")
|
|
126
|
+
self.assertEqual(init["result"]["serverInfo"]["name"], "iumbtems")
|
|
127
|
+
self.assertEqual(
|
|
128
|
+
sorted(t["name"] for t in tools_list["result"]["tools"]),
|
|
129
|
+
sorted(EXPECTED_TOOLS),
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
# Round-trip verify_quote in-process against the temp cache.
|
|
133
|
+
server = build_server()
|
|
134
|
+
res = server.handle({
|
|
135
|
+
"jsonrpc": "2.0",
|
|
136
|
+
"id": 3,
|
|
137
|
+
"method": "tools/call",
|
|
138
|
+
"params": {
|
|
139
|
+
"name": "iumbtems_verify_quote",
|
|
140
|
+
"arguments": {
|
|
141
|
+
"hash": digest,
|
|
142
|
+
"quote": "Poseidon in 184ms",
|
|
143
|
+
"base_dir": tmp,
|
|
144
|
+
},
|
|
145
|
+
},
|
|
146
|
+
})
|
|
147
|
+
payload = json.loads(res["result"]["content"][0]["text"])
|
|
148
|
+
self.assertTrue(payload["verified"])
|
|
149
|
+
self.assertGreaterEqual(payload["confidence"], 0.95)
|
|
150
|
+
|
|
151
|
+
# And a failing quote must come back unverified.
|
|
152
|
+
res_bad = server.handle({
|
|
153
|
+
"jsonrpc": "2.0",
|
|
154
|
+
"id": 4,
|
|
155
|
+
"method": "tools/call",
|
|
156
|
+
"params": {
|
|
157
|
+
"name": "iumbtems_verify_quote",
|
|
158
|
+
"arguments": {
|
|
159
|
+
"hash": digest,
|
|
160
|
+
"quote": "this text was never cached anywhere",
|
|
161
|
+
"base_dir": tmp,
|
|
162
|
+
},
|
|
163
|
+
},
|
|
164
|
+
})
|
|
165
|
+
bad = json.loads(res_bad["result"]["content"][0]["text"])
|
|
166
|
+
self.assertFalse(bad["verified"])
|
|
167
|
+
|
|
168
|
+
def test_one_shot_call_mode(self):
|
|
169
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
170
|
+
from skills.research_cache.hasher import SourceHasher
|
|
171
|
+
|
|
172
|
+
hasher = SourceHasher(Path(tmp))
|
|
173
|
+
content = "Epistemic sovereignty over parametric intuition."
|
|
174
|
+
digest = hasher.store_source(url="https://example.test/q", content=content)
|
|
175
|
+
|
|
176
|
+
proc = subprocess.run(
|
|
177
|
+
[
|
|
178
|
+
sys.executable,
|
|
179
|
+
str(PROJECT_ROOT / "runner" / "mcp_server.py"),
|
|
180
|
+
"call",
|
|
181
|
+
"iumbtems_verify_quote",
|
|
182
|
+
json.dumps({"hash": digest, "quote": "parametric intuition", "base_dir": tmp}),
|
|
183
|
+
],
|
|
184
|
+
capture_output=True,
|
|
185
|
+
text=True,
|
|
186
|
+
timeout=20,
|
|
187
|
+
)
|
|
188
|
+
self.assertEqual(proc.returncode, 0, proc.stderr)
|
|
189
|
+
payload = json.loads(proc.stdout)
|
|
190
|
+
self.assertTrue(payload["verified"])
|
|
191
|
+
|
|
192
|
+
def test_unknown_tool_reports_available(self):
|
|
193
|
+
proc = subprocess.run(
|
|
194
|
+
[
|
|
195
|
+
sys.executable,
|
|
196
|
+
str(PROJECT_ROOT / "runner" / "mcp_server.py"),
|
|
197
|
+
"call",
|
|
198
|
+
"nope_not_a_tool",
|
|
199
|
+
"{}",
|
|
200
|
+
],
|
|
201
|
+
capture_output=True,
|
|
202
|
+
text=True,
|
|
203
|
+
timeout=20,
|
|
204
|
+
)
|
|
205
|
+
self.assertEqual(proc.returncode, 1)
|
|
206
|
+
payload = json.loads(proc.stdout)
|
|
207
|
+
self.assertIn("error", payload)
|
|
208
|
+
self.assertEqual(sorted(payload["available"]), sorted(EXPECTED_TOOLS))
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
if __name__ == "__main__":
|
|
212
|
+
unittest.main()
|
|
@@ -0,0 +1,240 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""PCRB probe test (Stream D / Gate G3 falsification criterion).
|
|
3
|
+
|
|
4
|
+
[HYPOTHESIS: a standalone verifier over a 50-claim brief runs in <2s and
|
|
5
|
+
flags 100% of seeded quote tampering]
|
|
6
|
+
|
|
7
|
+
This test IS the probe: build a realistic 50-claim signed brief, time the
|
|
8
|
+
verifier, then seed >= 20 tampering mutations across four attack classes
|
|
9
|
+
(quote flips, source body edits, claim swaps, coordinated manifest rewrites)
|
|
10
|
+
and assert every single one is flagged.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import copy
|
|
14
|
+
import json
|
|
15
|
+
import sys
|
|
16
|
+
import tempfile
|
|
17
|
+
import time
|
|
18
|
+
import unittest
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
|
|
22
|
+
if str(PROJECT_ROOT) not in sys.path:
|
|
23
|
+
sys.path.insert(0, str(PROJECT_ROOT))
|
|
24
|
+
|
|
25
|
+
from runner.pcrb import export_brief, sha256_text, member_hash # noqa: E402
|
|
26
|
+
from runner.pcrb_verify import verify_brief # noqa: E402
|
|
27
|
+
|
|
28
|
+
TEST_KEY = "pcrb-probe-secret-key"
|
|
29
|
+
N_CLAIMS = 50
|
|
30
|
+
N_SOURCES = 5
|
|
31
|
+
VERIFY_BUDGET_S = 2.0
|
|
32
|
+
N_MUTATIONS = 20
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def build_workspace(base: Path) -> dict:
|
|
36
|
+
"""A .research-shaped workspace: 5 sources x 10 claims, all quotes real."""
|
|
37
|
+
sources = base / "sources"
|
|
38
|
+
sources.mkdir(parents=True, exist_ok=True)
|
|
39
|
+
scratch = base / "scratchpads" / "scope_01"
|
|
40
|
+
scratch.mkdir(parents=True, exist_ok=True)
|
|
41
|
+
|
|
42
|
+
src_hashes = []
|
|
43
|
+
for s in range(N_SOURCES):
|
|
44
|
+
content = (
|
|
45
|
+
f"BENCHMARK REPORT {s}: The prover executes witness generation in "
|
|
46
|
+
f"{180 + s}ms under load, with peak bandwidth {40 + s} GB/s and "
|
|
47
|
+
f"validated reproducibility across {3 + s} independent runs. "
|
|
48
|
+
f"Section {s} documents the methodology and error bars in full."
|
|
49
|
+
)
|
|
50
|
+
h = sha256_text(content)
|
|
51
|
+
(sources / f"{h}.md").write_text(content, encoding="utf-8")
|
|
52
|
+
(sources / f"{h}.json").write_text(
|
|
53
|
+
json.dumps({"hash": h, "url": f"https://bench.example/{s}", "title": f"Bench {s}", "tier": "WEB_DOCUMENT"})
|
|
54
|
+
)
|
|
55
|
+
src_hashes.append(h)
|
|
56
|
+
|
|
57
|
+
alpha_claims, beta_claims = [], []
|
|
58
|
+
for i in range(N_CLAIMS):
|
|
59
|
+
s = i % N_SOURCES
|
|
60
|
+
content = (sources / f"{src_hashes[s]}.md").read_text(encoding="utf-8")
|
|
61
|
+
# Unique verbatim substring from that source.
|
|
62
|
+
quote = f"validated reproducibility across {3 + s} independent runs"
|
|
63
|
+
claim = {
|
|
64
|
+
"claim_id": f"C-{i:03d}",
|
|
65
|
+
"tag": "VERIFIED",
|
|
66
|
+
"statement": f"Claim {i}: prover behavior measured in benchmark {s}",
|
|
67
|
+
"source_hash": src_hashes[s],
|
|
68
|
+
"source_url": f"https://bench.example/{s}",
|
|
69
|
+
"verbatim_quote": quote,
|
|
70
|
+
}
|
|
71
|
+
if i % 2 == 0:
|
|
72
|
+
alpha_claims.append(claim)
|
|
73
|
+
else:
|
|
74
|
+
claim["severity"] = "LOW"
|
|
75
|
+
beta_claims.append(claim)
|
|
76
|
+
|
|
77
|
+
(scratch / "alpha_dossier.json").write_text(
|
|
78
|
+
json.dumps({"agent": "Agent Alpha", "scope_id": "scope_01", "affirmative_claims": alpha_claims, "inferred_implications": [], "negative_knowledge": []})
|
|
79
|
+
)
|
|
80
|
+
(scratch / "beta_dossier.json").write_text(
|
|
81
|
+
json.dumps({"agent": "Agent Beta", "scope_id": "scope_01", "falsification_claims": beta_claims, "methodological_critiques": [], "negative_knowledge": []})
|
|
82
|
+
)
|
|
83
|
+
(base / "final_synthesis.md").write_text("# Master Synthesis\n\nAll claims verified against benchmark sources.\n")
|
|
84
|
+
return {"src_hashes": src_hashes}
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class TestPcrbProbe(unittest.TestCase):
|
|
88
|
+
@classmethod
|
|
89
|
+
def setUpClass(cls):
|
|
90
|
+
cls._tmp = tempfile.TemporaryDirectory(prefix="pcrb_probe_")
|
|
91
|
+
cls.base = Path(cls._tmp.name)
|
|
92
|
+
cls.fx = build_workspace(cls.base)
|
|
93
|
+
cls.bundle_path = export_brief(cls.base, key=TEST_KEY, objective="PCRB probe")
|
|
94
|
+
cls.bundle = json.loads(cls.bundle_path.read_text(encoding="utf-8"))
|
|
95
|
+
|
|
96
|
+
@classmethod
|
|
97
|
+
def tearDownClass(cls):
|
|
98
|
+
cls._tmp.cleanup()
|
|
99
|
+
|
|
100
|
+
def test_bundle_shape(self):
|
|
101
|
+
self.assertEqual(self.bundle["pcrb_version"], "1.0")
|
|
102
|
+
self.assertEqual(len(self.bundle["claims"]), N_CLAIMS)
|
|
103
|
+
self.assertEqual(len(self.bundle["sources"]), N_SOURCES)
|
|
104
|
+
self.assertEqual(self.bundle["signature"]["alg"], "hmac-sha256")
|
|
105
|
+
self.assertIn("manifest", self.bundle)
|
|
106
|
+
|
|
107
|
+
def test_probe_clean_brief_verifies_under_two_seconds(self):
|
|
108
|
+
"""THE probe (half 1): 50-claim brief verifies in <2s."""
|
|
109
|
+
key = TEST_KEY.encode("utf-8")
|
|
110
|
+
started = time.perf_counter()
|
|
111
|
+
report = verify_brief(self.bundle, key=key)
|
|
112
|
+
elapsed = time.perf_counter() - started
|
|
113
|
+
|
|
114
|
+
self.assertTrue(report["ok"], report["failures"])
|
|
115
|
+
self.assertEqual(report["stats"]["quotes_verified"], N_CLAIMS)
|
|
116
|
+
self.assertEqual(report["stats"]["quotes_failed"], 0)
|
|
117
|
+
self.assertLess(elapsed, VERIFY_BUDGET_S, f"verify took {elapsed:.3f}s")
|
|
118
|
+
self.assertLess(report["stats"]["elapsed_s"], VERIFY_BUDGET_S)
|
|
119
|
+
|
|
120
|
+
def test_probe_twenty_tamper_mutations_all_flagged(self):
|
|
121
|
+
"""THE probe (half 2): 100% of >=20 seeded mutations are caught."""
|
|
122
|
+
key = TEST_KEY.encode("utf-8")
|
|
123
|
+
mutations = self._seed_mutations()
|
|
124
|
+
self.assertGreaterEqual(len(mutations), N_MUTATIONS)
|
|
125
|
+
|
|
126
|
+
caught = []
|
|
127
|
+
missed = []
|
|
128
|
+
for name, tampered in mutations:
|
|
129
|
+
report = verify_brief(tampered, key=key)
|
|
130
|
+
if report["ok"]:
|
|
131
|
+
missed.append(name)
|
|
132
|
+
else:
|
|
133
|
+
caught.append((name, [f["check"] for f in report["failures"]]))
|
|
134
|
+
|
|
135
|
+
self.assertEqual(
|
|
136
|
+
missed, [],
|
|
137
|
+
f"MISSED tampering ({len(missed)}/{len(mutations)}): {missed}",
|
|
138
|
+
)
|
|
139
|
+
self.assertEqual(len(caught), len(mutations), "every mutation must be flagged")
|
|
140
|
+
|
|
141
|
+
def _seed_mutations(self):
|
|
142
|
+
"""Four attack classes; each mutation is a fresh deep copy."""
|
|
143
|
+
out = []
|
|
144
|
+
|
|
145
|
+
# Class 1: quote character flips inside claim records (5x)
|
|
146
|
+
for i in range(5):
|
|
147
|
+
b = copy.deepcopy(self.bundle)
|
|
148
|
+
rec = b["claims"][i]
|
|
149
|
+
q = rec["verbatim_quote"]
|
|
150
|
+
rec["verbatim_quote"] = q.replace("reproducibility", "reproducib1lity", 1)
|
|
151
|
+
out.append((f"quote_flip_{i}", b))
|
|
152
|
+
|
|
153
|
+
# Class 2: source body edits (5x)
|
|
154
|
+
for i in range(5):
|
|
155
|
+
b = copy.deepcopy(self.bundle)
|
|
156
|
+
src = self.fx["src_hashes"][i % N_SOURCES]
|
|
157
|
+
b["sources"][src]["content"] = b["sources"][src]["content"] + "\nINJECTED FALSE PARAGRAPH."
|
|
158
|
+
out.append((f"source_body_edit_{i}", b))
|
|
159
|
+
|
|
160
|
+
# Class 3: claim swaps — replace a claim wholesale with a different one (5x)
|
|
161
|
+
for i in range(5):
|
|
162
|
+
b = copy.deepcopy(self.bundle)
|
|
163
|
+
donor = copy.deepcopy(b["claims"][(i + 7) % N_CLAIMS])
|
|
164
|
+
donor["claim_id"] = b["claims"][i]["claim_id"] # same id, different content
|
|
165
|
+
b["claims"][i] = donor
|
|
166
|
+
out.append((f"claim_swap_{i}", b))
|
|
167
|
+
|
|
168
|
+
# Class 4: coordinated rewrites — attacker fixes the manifest hash too
|
|
169
|
+
# (only the HMAC signature can catch these) (3x claims + 2x sources)
|
|
170
|
+
for i in range(3):
|
|
171
|
+
b = copy.deepcopy(self.bundle)
|
|
172
|
+
rec = b["claims"][i + 10]
|
|
173
|
+
rec["statement"] = rec["statement"] + " (fabricated conclusion)"
|
|
174
|
+
rec["verbatim_quote"] = rec["verbatim_quote"].replace("independent", "independant", 1)
|
|
175
|
+
b["manifest"]["claims"][rec["claim_id"]] = member_hash(rec) # attacker "fixes" the hash
|
|
176
|
+
out.append((f"coordinated_claim_rewrite_{i}", b))
|
|
177
|
+
|
|
178
|
+
for i in range(2):
|
|
179
|
+
b = copy.deepcopy(self.bundle)
|
|
180
|
+
src = self.fx["src_hashes"][i]
|
|
181
|
+
b["sources"][src]["content"] = b["sources"][src]["content"].replace("GB/s", "TB/s", 1)
|
|
182
|
+
b["manifest"]["sources"][src] = sha256_text(b["sources"][src]["content"]) # attacker "fixes" it
|
|
183
|
+
out.append((f"coordinated_source_rewrite_{i}", b))
|
|
184
|
+
|
|
185
|
+
return out
|
|
186
|
+
|
|
187
|
+
def test_unsigned_mode_still_catches_member_mismatches(self):
|
|
188
|
+
"""alg=none: integrity catches manifest mismatches; no authenticity claim."""
|
|
189
|
+
unsigned = export_brief(self.base, out_path=self.base / "unsigned.pcrb.json", objective="unsigned probe")
|
|
190
|
+
bundle = json.loads(unsigned.read_text(encoding="utf-8"))
|
|
191
|
+
self.assertEqual(bundle["signature"]["alg"], "none")
|
|
192
|
+
|
|
193
|
+
report = verify_brief(bundle, key=None)
|
|
194
|
+
self.assertTrue(report["ok"], report["failures"])
|
|
195
|
+
self.assertEqual(report["checks"]["signature"], "unsigned")
|
|
196
|
+
|
|
197
|
+
# Member tamper still caught without a key...
|
|
198
|
+
tampered = copy.deepcopy(bundle)
|
|
199
|
+
tampered["claims"][0]["verbatim_quote"] = "totally fabricated quote text"
|
|
200
|
+
report2 = verify_brief(tampered, key=None)
|
|
201
|
+
self.assertFalse(report2["ok"])
|
|
202
|
+
|
|
203
|
+
# ...but a COORDINATED rewrite (content + manifest) is undetectable
|
|
204
|
+
# without a key. This is the honest limitation of alg=none.
|
|
205
|
+
coordinated = copy.deepcopy(bundle)
|
|
206
|
+
coordinated["claims"][0]["verbatim_quote"] = "totally fabricated quote text"
|
|
207
|
+
coordinated["manifest"]["claims"][coordinated["claims"][0]["claim_id"]] = member_hash(
|
|
208
|
+
coordinated["claims"][0]
|
|
209
|
+
)
|
|
210
|
+
report3 = verify_brief(coordinated, key=None)
|
|
211
|
+
self.assertTrue(
|
|
212
|
+
report3["ok"],
|
|
213
|
+
"alg=none cannot detect coordinated rewrites — documented limitation, not a bug",
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
def test_missing_key_is_reported_not_silently_passed(self):
|
|
217
|
+
report = verify_brief(self.bundle, key=None)
|
|
218
|
+
self.assertFalse(report["ok"])
|
|
219
|
+
self.assertEqual(report["checks"]["signature"], "unverifiable")
|
|
220
|
+
|
|
221
|
+
def test_wrong_key_fails(self):
|
|
222
|
+
report = verify_brief(self.bundle, key=b"wrong-key")
|
|
223
|
+
self.assertFalse(report["ok"])
|
|
224
|
+
self.assertEqual(report["checks"]["signature"], "invalid")
|
|
225
|
+
|
|
226
|
+
def test_cli_exit_codes(self):
|
|
227
|
+
import subprocess
|
|
228
|
+
|
|
229
|
+
proc = subprocess.run(
|
|
230
|
+
[sys.executable, str(PROJECT_ROOT / "runner" / "pcrb_verify.py"), str(self.bundle_path), "--json"],
|
|
231
|
+
capture_output=True, text=True, timeout=30,
|
|
232
|
+
env={**__import__("os").environ, "IUMBTEMS_PCRB_KEY": TEST_KEY},
|
|
233
|
+
)
|
|
234
|
+
self.assertEqual(proc.returncode, 0, proc.stdout + proc.stderr)
|
|
235
|
+
data = json.loads(proc.stdout)
|
|
236
|
+
self.assertTrue(data["ok"])
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
if __name__ == "__main__":
|
|
240
|
+
unittest.main()
|