@heretek-ai/epistemic-swarm 0.2.2 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/skills/brainstorming/SKILL.md +13 -0
- package/.agents/skills/code_audit/SKILL.md +13 -0
- package/.agents/skills/epistemic_search/SKILL.md +13 -0
- package/.agents/skills/grilling/SKILL.md +13 -0
- package/.agents/skills/oss_scout/SKILL.md +13 -0
- package/.agents/skills/research_cache/SKILL.md +13 -0
- package/.agents/skills/swarm_config/SKILL.md +13 -0
- package/.claude-plugin/plugin.json +15 -5
- package/.omp/README.md +39 -0
- package/.omp/SYSTEM.md +12 -0
- package/.omp/commands/audit.md +10 -0
- package/.omp/commands/brainstorming.md +12 -0
- package/.omp/commands/grill.md +9 -0
- package/.omp/commands/scout.md +10 -0
- package/.omp/commands/swarm-config.md +10 -0
- package/.omp/commands/swarm.md +9 -0
- package/.omp/hooks/post/epistemic-audit.ts +16 -0
- package/.omp/hooks/pre/epistemic-redirect.ts +21 -0
- package/.omp/prompts/brainstorming.md +8 -0
- package/.omp/prompts/swarm.md +8 -0
- package/MARKETPLACE.md +8 -0
- package/README.md +53 -8
- package/bin/cli.js +27 -3
- package/config/domain_packs/biopharma.json +23 -0
- package/config/domain_packs/legal.json +19 -0
- package/config/domain_packs/quant.json +19 -0
- package/config/mcp-research-servers.json +7 -0
- package/config/mcp_launcher.py +48 -137
- package/config/opencode-snippet.json +58 -3
- package/config/searxng_mcp.py +42 -83
- package/extensions/pi/index.js +196 -28
- package/install.sh +20 -4
- package/package.json +40 -5
- package/plugins/antigravity/README.md +28 -0
- package/plugins/antigravity/agents/alpha-thesis.md +6 -0
- package/plugins/antigravity/agents/beta-antithesis.md +7 -0
- package/plugins/antigravity/agents/brainstormer.md +7 -0
- package/plugins/antigravity/agents/epistemic-auditor.md +5 -0
- package/plugins/antigravity/hooks.json +23 -0
- package/plugins/antigravity/mcp_config.json +33 -0
- package/plugins/antigravity/plugin.json +21 -0
- package/plugins/antigravity/rules/epistemic-integrity.md +6 -0
- package/plugins/antigravity/skills/brainstorming/SKILL.md +13 -0
- package/plugins/antigravity/skills/code_audit/SKILL.md +13 -0
- package/plugins/antigravity/skills/epistemic_search/SKILL.md +13 -0
- package/plugins/antigravity/skills/grilling/SKILL.md +13 -0
- package/plugins/antigravity/skills/oss_scout/SKILL.md +13 -0
- package/plugins/antigravity/skills/research_cache/SKILL.md +13 -0
- package/plugins/antigravity/skills/swarm_config/SKILL.md +13 -0
- package/plugins/codex/AGENTS.md.snippet +10 -0
- package/plugins/codex/README.md +37 -0
- package/plugins/codex/config.toml.snippet +28 -0
- package/plugins/codex/openai.yaml +24 -0
- package/plugins/codex/skills/brainstorming/SKILL.md +13 -0
- package/plugins/codex/skills/code_audit/SKILL.md +13 -0
- package/plugins/codex/skills/epistemic_search/SKILL.md +13 -0
- package/plugins/codex/skills/grilling/SKILL.md +13 -0
- package/plugins/codex/skills/oss_scout/SKILL.md +13 -0
- package/plugins/codex/skills/research_cache/SKILL.md +13 -0
- package/plugins/codex/skills/swarm_config/SKILL.md +13 -0
- package/plugins/gemini/GEMINI.md +15 -0
- package/plugins/gemini/README.md +19 -0
- package/plugins/gemini/commands/audit.toml +6 -0
- package/plugins/gemini/commands/brainstorming.toml +10 -0
- package/plugins/gemini/commands/grill.toml +6 -0
- package/plugins/gemini/commands/scout.toml +7 -0
- package/plugins/gemini/commands/swarm-config.toml +7 -0
- package/plugins/gemini/commands/swarm.toml +8 -0
- package/plugins/gemini/gemini-extension.json +38 -0
- package/plugins/gemini/hooks/hooks.json +11 -0
- package/plugins/gemini/skills/brainstorming/SKILL.md +13 -0
- package/plugins/gemini/skills/code_audit/SKILL.md +13 -0
- package/plugins/gemini/skills/epistemic_search/SKILL.md +13 -0
- package/plugins/gemini/skills/grilling/SKILL.md +13 -0
- package/plugins/gemini/skills/oss_scout/SKILL.md +13 -0
- package/plugins/gemini/skills/research_cache/SKILL.md +13 -0
- package/plugins/gemini/skills/swarm_config/SKILL.md +13 -0
- package/plugins/opencode/index.js +335 -118
- package/prompts/agent_brainstormer.md +97 -0
- package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
- package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
- package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
- package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
- package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
- package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
- package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
- package/runner/auctioneer.py +169 -0
- package/runner/auditor_engine.py +127 -12
- package/runner/claim_store.py +361 -0
- package/runner/claim_witness.py +183 -0
- package/runner/living_dossiers.py +355 -0
- package/runner/mcp_protocol.py +188 -0
- package/runner/mcp_server.py +567 -0
- package/runner/pcrb.py +212 -0
- package/runner/pcrb_verify.py +237 -0
- package/runner/refinement.py +335 -0
- package/runner/research_swarm.py +378 -94
- package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
- package/runner/tests/fixtures/auction_objective.json +35 -0
- package/runner/tests/fixtures/borderline_claims.json +27 -0
- package/runner/tests/fixtures/divergence_objectives.json +31 -0
- package/runner/tests/test_auction_order.py +173 -0
- package/runner/tests/test_claim_store.py +162 -0
- package/runner/tests/test_claim_witness.py +186 -0
- package/runner/tests/test_domain_packs.py +217 -0
- package/runner/tests/test_fleet_seam.py +113 -0
- package/runner/tests/test_living_dossiers.py +204 -0
- package/runner/tests/test_mcp_server.py +212 -0
- package/runner/tests/test_pcrb.py +240 -0
- package/runner/tests/test_refinement.py +255 -0
- package/runner/tests/test_swarm.py +173 -15
- package/scripts/auction_experiment.py +180 -0
- package/scripts/build_adapters.py +183 -0
- package/scripts/divergence_experiment.py +184 -0
- package/skills/brainstorming/SKILL.md +106 -0
- package/skills/brainstorming/__init__.py +1 -0
- package/skills/brainstorming/scripts/brainstorm.py +200 -0
- package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
- package/skills/swarm_config/SKILL.md +1 -1
- package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
- package/skills/swarm_config/configure.py +45 -11
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Regulated Domain Packs probe test (Stream G falsification criterion).
|
|
3
|
+
|
|
4
|
+
[HYPOTHESIS: a biopharma constitution flips accept/reject on >= 25% of a
|
|
5
|
+
20-claim borderline fixture set]
|
|
6
|
+
|
|
7
|
+
This test IS the probe: the 20-claim borderline fixture is judged under the
|
|
8
|
+
legacy default constitution and under the biopharma pack; >= 25% of per-claim
|
|
9
|
+
verdicts must flip. The zero-tolerance retraction policy must downgrade
|
|
10
|
+
claims with retracted sources regardless of score.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
import sys
|
|
15
|
+
import unittest
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
|
|
19
|
+
if str(PROJECT_ROOT) not in sys.path:
|
|
20
|
+
sys.path.insert(0, str(PROJECT_ROOT))
|
|
21
|
+
|
|
22
|
+
from runner.claim_witness import normalize_claim # noqa: E402
|
|
23
|
+
from runner.refinement import ( # noqa: E402
|
|
24
|
+
LEGACY_CONSTITUTION,
|
|
25
|
+
claim_verdict,
|
|
26
|
+
compute_epistemic_score_from_claims,
|
|
27
|
+
load_domain_pack,
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
FIXTURE = PROJECT_ROOT / "runner" / "tests" / "fixtures" / "borderline_claims.json"
|
|
31
|
+
N_CLAIMS = 20
|
|
32
|
+
FLIP_BAR = 0.25
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def load_fixture():
|
|
36
|
+
data = json.loads(FIXTURE.read_text(encoding="utf-8"))
|
|
37
|
+
claims = [normalize_claim(c) for c in data["claims"]]
|
|
38
|
+
return claims, set(data["retracted_hashes"])
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class TestDomainPacksProbe(unittest.TestCase):
|
|
42
|
+
@classmethod
|
|
43
|
+
def setUpClass(cls):
|
|
44
|
+
cls.claims, cls.retracted = load_fixture()
|
|
45
|
+
cls.biopharma = load_domain_pack("biopharma")
|
|
46
|
+
|
|
47
|
+
def test_fixture_shape(self):
|
|
48
|
+
self.assertEqual(len(self.claims), N_CLAIMS)
|
|
49
|
+
|
|
50
|
+
def test_load_all_three_packs(self):
|
|
51
|
+
for pack_id in ("biopharma", "quant", "legal"):
|
|
52
|
+
c = load_domain_pack(pack_id)
|
|
53
|
+
self.assertIsNotNone(c.tier_weights)
|
|
54
|
+
self.assertIn("PREPRINT", c.tier_weights)
|
|
55
|
+
with self.assertRaises(FileNotFoundError):
|
|
56
|
+
load_domain_pack("no-such-pack")
|
|
57
|
+
|
|
58
|
+
def test_probe_biopharma_flips_at_least_quarter(self):
|
|
59
|
+
"""THE probe: >= 25% of 20 borderline verdicts flip under biopharma."""
|
|
60
|
+
default_results = [
|
|
61
|
+
claim_verdict(c, constitution=LEGACY_CONSTITUTION, retracted_hashes=self.retracted)
|
|
62
|
+
for c in self.claims
|
|
63
|
+
]
|
|
64
|
+
bio_results = [
|
|
65
|
+
claim_verdict(c, constitution=self.biopharma, retracted_hashes=self.retracted)
|
|
66
|
+
for c in self.claims
|
|
67
|
+
]
|
|
68
|
+
|
|
69
|
+
flips = [
|
|
70
|
+
(d["claim_id"], d["verdict"], b["verdict"])
|
|
71
|
+
for d, b in zip(default_results, bio_results)
|
|
72
|
+
if d["verdict"] != b["verdict"]
|
|
73
|
+
]
|
|
74
|
+
flip_rate = len(flips) / N_CLAIMS
|
|
75
|
+
self.assertGreaterEqual(
|
|
76
|
+
flip_rate,
|
|
77
|
+
FLIP_BAR,
|
|
78
|
+
f"flip rate {flip_rate:.0%} ({len(flips)}/{N_CLAIMS}) below {FLIP_BAR:.0%}: {flips}",
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
# Default constitution accepts the whole borderline set...
|
|
82
|
+
self.assertTrue(all(r["verdict"] == "ACCEPTED" for r in default_results),
|
|
83
|
+
[r for r in default_results if r["verdict"] != "ACCEPTED"])
|
|
84
|
+
# ...and every flip is default-ACCEPTED -> biopharma-REJECTED.
|
|
85
|
+
self.assertTrue(all(d == "ACCEPTED" and b == "REJECTED" for _c, d, b in flips))
|
|
86
|
+
|
|
87
|
+
def test_banned_domain_mechanism(self):
|
|
88
|
+
banned = [c for c in self.claims if "mirror.example" in (c.source_url or "") or "predatory-journal" in (c.source_url or "")]
|
|
89
|
+
self.assertEqual(len(banned), 6, "fixture expects 6 banned-domain claims")
|
|
90
|
+
for c in banned:
|
|
91
|
+
r = claim_verdict(c, constitution=self.biopharma)
|
|
92
|
+
self.assertEqual(r["verdict"], "REJECTED", c.claim_id)
|
|
93
|
+
self.assertTrue(any("BANNED_DOMAIN" in x for x in r["reasons"]), r["reasons"])
|
|
94
|
+
|
|
95
|
+
def test_missing_tag_mechanism(self):
|
|
96
|
+
untagged = [c for c in self.claims if not c.tag]
|
|
97
|
+
self.assertEqual(len(untagged), 2)
|
|
98
|
+
for c in untagged:
|
|
99
|
+
r = claim_verdict(c, constitution=self.biopharma)
|
|
100
|
+
self.assertEqual(r["verdict"], "REJECTED", c.claim_id)
|
|
101
|
+
self.assertTrue(any("MISSING_TAG" in x for x in r["reasons"]), r["reasons"])
|
|
102
|
+
|
|
103
|
+
def test_zero_tolerance_retraction_mechanism(self):
|
|
104
|
+
"""RETRACTED source downgrades regardless of quote match or score."""
|
|
105
|
+
retracted_claims = [c for c in self.claims if c.source_hash in self.retracted]
|
|
106
|
+
self.assertEqual(len(retracted_claims), 2)
|
|
107
|
+
for c in retracted_claims:
|
|
108
|
+
# Even without passing retracted_hashes: status-based downgrade.
|
|
109
|
+
c.status = "STALE"
|
|
110
|
+
r = claim_verdict(c, constitution=self.biopharma)
|
|
111
|
+
self.assertEqual(r["verdict"], "REJECTED", c.claim_id)
|
|
112
|
+
self.assertTrue(
|
|
113
|
+
any("ZERO_TOLERANCE_RETRACTION" in x for x in r["reasons"]), r["reasons"]
|
|
114
|
+
)
|
|
115
|
+
# Standard policy does NOT apply this rule.
|
|
116
|
+
r_std = claim_verdict(c, constitution=LEGACY_CONSTITUTION)
|
|
117
|
+
self.assertEqual(r_std["verdict"], "ACCEPTED")
|
|
118
|
+
|
|
119
|
+
def test_tier_weights_move_the_score(self):
|
|
120
|
+
"""Preprint downweighting lowers E(D); threshold also comes from the pack."""
|
|
121
|
+
s_default, _ = compute_epistemic_score_from_claims(self.claims, constitution=LEGACY_CONSTITUTION)
|
|
122
|
+
s_bio, br = compute_epistemic_score_from_claims(self.claims, constitution=self.biopharma)
|
|
123
|
+
self.assertLess(s_bio, s_default, (s_default, s_bio))
|
|
124
|
+
# 18 of 20 claims carry the VERIFIED tag (2 are deliberately untagged);
|
|
125
|
+
# each scores 0.3 under the biopharma preprint weight.
|
|
126
|
+
n_verified = sum(1 for c in self.claims if c.tag == "VERIFIED")
|
|
127
|
+
self.assertEqual(n_verified, 18)
|
|
128
|
+
self.assertAlmostEqual(br["verified_weight"], n_verified * 0.3, places=3)
|
|
129
|
+
|
|
130
|
+
def test_legacy_default_preserves_existing_behavior(self):
|
|
131
|
+
"""No pack => flat weights, no domain/tag rules, standard retraction."""
|
|
132
|
+
self.assertEqual(LEGACY_CONSTITUTION.weight_for("PREPRINT"), 1.0)
|
|
133
|
+
self.assertEqual(LEGACY_CONSTITUTION.banned_domains, [])
|
|
134
|
+
self.assertEqual(LEGACY_CONSTITUTION.mandatory_tags, [])
|
|
135
|
+
self.assertEqual(LEGACY_CONSTITUTION.retraction_policy, "standard")
|
|
136
|
+
self.assertEqual(LEGACY_CONSTITUTION.accept_threshold, 0.65)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class TestAuditConstitutionWiring(unittest.TestCase):
|
|
140
|
+
def test_audit_scope_default_unchanged(self):
|
|
141
|
+
"""audit_scope() with no constitution must match legacy numbers exactly."""
|
|
142
|
+
import tempfile
|
|
143
|
+
from runner.auditor_engine import EpistemicAuditorEngine
|
|
144
|
+
from runner.state_machine import ResearchStateMachine
|
|
145
|
+
from skills.research_cache.hasher import SourceHasher
|
|
146
|
+
|
|
147
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
148
|
+
base = Path(tmp)
|
|
149
|
+
hasher = SourceHasher(base)
|
|
150
|
+
sm = ResearchStateMachine(base_dir=base)
|
|
151
|
+
shash = hasher.store_source(
|
|
152
|
+
"https://x", "FPGA prover executes Poseidon in 184ms.", "T"
|
|
153
|
+
)
|
|
154
|
+
sm.init_session("t")
|
|
155
|
+
sm.set_scopes([{"scope_id": "s1", "dependencies": []}])
|
|
156
|
+
sm.record_agent_completion("s1", "alpha", {
|
|
157
|
+
"affirmative_claims": [{
|
|
158
|
+
"claim_id": "A1", "tag": "VERIFIED", "statement": "s",
|
|
159
|
+
"source_hash": shash, "verbatim_quote": "executes Poseidon in 184ms",
|
|
160
|
+
}],
|
|
161
|
+
"inferred_implications": [],
|
|
162
|
+
"negative_knowledge": [],
|
|
163
|
+
})
|
|
164
|
+
sm.record_agent_completion("s1", "beta", {
|
|
165
|
+
"falsification_claims": [],
|
|
166
|
+
"methodological_critiques": [],
|
|
167
|
+
"negative_knowledge": [],
|
|
168
|
+
})
|
|
169
|
+
|
|
170
|
+
engine = EpistemicAuditorEngine(base_dir=base)
|
|
171
|
+
report = engine.audit_scope("s1")
|
|
172
|
+
self.assertEqual(report["summary"]["verified_passed"], 1)
|
|
173
|
+
self.assertEqual(report["summary"]["verdict"], "CERTIFIED")
|
|
174
|
+
|
|
175
|
+
def test_audit_scope_with_biopharma_uses_threshold(self):
|
|
176
|
+
import tempfile
|
|
177
|
+
from runner.auditor_engine import EpistemicAuditorEngine
|
|
178
|
+
from runner.refinement import load_domain_pack
|
|
179
|
+
from runner.state_machine import ResearchStateMachine
|
|
180
|
+
from skills.research_cache.hasher import SourceHasher
|
|
181
|
+
|
|
182
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
183
|
+
base = Path(tmp)
|
|
184
|
+
hasher = SourceHasher(base)
|
|
185
|
+
sm = ResearchStateMachine(base_dir=base)
|
|
186
|
+
shash = hasher.store_source(
|
|
187
|
+
"https://predatory-journal.example/x",
|
|
188
|
+
"Weak preprint finding about surrogate endpoints.",
|
|
189
|
+
"T", tier="PREPRINT",
|
|
190
|
+
)
|
|
191
|
+
sm.init_session("t")
|
|
192
|
+
sm.set_scopes([{"scope_id": "s1", "dependencies": []}])
|
|
193
|
+
sm.record_agent_completion("s1", "alpha", {
|
|
194
|
+
"affirmative_claims": [{
|
|
195
|
+
"claim_id": "A1", "tag": "VERIFIED", "statement": "s",
|
|
196
|
+
"source_hash": shash,
|
|
197
|
+
"source_url": "https://predatory-journal.example/x",
|
|
198
|
+
"verbatim_quote": "surrogate endpoints",
|
|
199
|
+
}],
|
|
200
|
+
"inferred_implications": [],
|
|
201
|
+
"negative_knowledge": [],
|
|
202
|
+
})
|
|
203
|
+
sm.record_agent_completion("s1", "beta", {
|
|
204
|
+
"falsification_claims": [],
|
|
205
|
+
"methodological_critiques": [],
|
|
206
|
+
"negative_knowledge": [],
|
|
207
|
+
})
|
|
208
|
+
|
|
209
|
+
engine = EpistemicAuditorEngine(base_dir=base)
|
|
210
|
+
report = engine.audit_scope("s1", constitution=load_domain_pack("biopharma"))
|
|
211
|
+
# Banned domain claim => REJECTED under biopharma.
|
|
212
|
+
self.assertEqual(report["summary"]["unverified_rejected"], 1)
|
|
213
|
+
self.assertEqual(report["summary"]["verified_passed"], 0)
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
if __name__ == "__main__":
|
|
217
|
+
unittest.main()
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Tests for the Stream E model-diverse adversary fleet seam.
|
|
3
|
+
|
|
4
|
+
The seam is config-first (per-agent backend/model), CLI-overridable, with env
|
|
5
|
+
fallback. Mock mode must NEVER build or spawn these commands.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import os
|
|
9
|
+
import sys
|
|
10
|
+
import unittest
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from unittest import mock
|
|
13
|
+
|
|
14
|
+
PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
|
|
15
|
+
if str(PROJECT_ROOT) not in sys.path:
|
|
16
|
+
sys.path.insert(0, str(PROJECT_ROOT))
|
|
17
|
+
|
|
18
|
+
from runner.research_swarm import SwarmRunner # noqa: E402
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class TestFleetSeam(unittest.TestCase):
|
|
22
|
+
def test_default_backend_is_claude_with_no_model(self):
|
|
23
|
+
runner = SwarmRunner(mock_mode=True, base_dir=Path("/tmp/no-such-research-dir"))
|
|
24
|
+
cmd = runner.build_agent_cmd("prompt", role="alpha")
|
|
25
|
+
self.assertEqual(cmd[0:2], ["claude", "-p"])
|
|
26
|
+
self.assertNotIn("--model", cmd)
|
|
27
|
+
self.assertIn("prompt", cmd)
|
|
28
|
+
|
|
29
|
+
def test_model_config_adds_model_flag(self):
|
|
30
|
+
runner = SwarmRunner(
|
|
31
|
+
mock_mode=True,
|
|
32
|
+
base_dir=Path("/tmp/no-such-research-dir"),
|
|
33
|
+
agent_overrides={"alpha": {"model": "claude-opus-5"}},
|
|
34
|
+
)
|
|
35
|
+
cmd = runner.build_agent_cmd("prompt", role="alpha")
|
|
36
|
+
self.assertIn("--model", cmd)
|
|
37
|
+
self.assertEqual(cmd[cmd.index("--model") + 1], "claude-opus-5")
|
|
38
|
+
|
|
39
|
+
def test_beta_backend_command_list(self):
|
|
40
|
+
"""Cross-FAMILY means non-Claude processes; backend is a cmd list."""
|
|
41
|
+
runner = SwarmRunner(
|
|
42
|
+
mock_mode=True,
|
|
43
|
+
base_dir=Path("/tmp/no-such-research-dir"),
|
|
44
|
+
agent_overrides={"beta": {"backend": ["ollama", "run", "qwen3"], "model": None}},
|
|
45
|
+
)
|
|
46
|
+
cmd = runner.build_agent_cmd("prompt", role="beta")
|
|
47
|
+
self.assertEqual(cmd[:3], ["ollama", "run", "qwen3"])
|
|
48
|
+
self.assertNotIn("--model", cmd)
|
|
49
|
+
|
|
50
|
+
def test_roles_resolve_independently(self):
|
|
51
|
+
"""Asymmetry is the point: alpha and beta may use different families."""
|
|
52
|
+
runner = SwarmRunner(
|
|
53
|
+
mock_mode=True,
|
|
54
|
+
base_dir=Path("/tmp/no-such-research-dir"),
|
|
55
|
+
agent_overrides={
|
|
56
|
+
"alpha": {"model": "claude-opus-5"},
|
|
57
|
+
"beta": {"backend": ["codex", "exec"], "model": "gpt-5"},
|
|
58
|
+
},
|
|
59
|
+
)
|
|
60
|
+
cmd_a = runner.build_agent_cmd("p", role="alpha")
|
|
61
|
+
cmd_b = runner.build_agent_cmd("p", role="beta")
|
|
62
|
+
self.assertEqual(cmd_a[0:2], ["claude", "-p"])
|
|
63
|
+
self.assertIn("--model", cmd_a)
|
|
64
|
+
self.assertEqual(cmd_b[:2], ["codex", "exec"])
|
|
65
|
+
self.assertIn("--model", cmd_b)
|
|
66
|
+
self.assertEqual(cmd_b[cmd_b.index("--model") + 1], "gpt-5")
|
|
67
|
+
|
|
68
|
+
def test_env_fallback(self):
|
|
69
|
+
with mock.patch.dict(
|
|
70
|
+
os.environ,
|
|
71
|
+
{"IUMBTEMS_MODEL_ALPHA": "env-model", "IUMBTEMS_BACKEND_BETA": "gemini -p"},
|
|
72
|
+
clear=False,
|
|
73
|
+
):
|
|
74
|
+
runner = SwarmRunner(mock_mode=True, base_dir=Path("/tmp/no-such-research-dir"))
|
|
75
|
+
cmd_a = runner.build_agent_cmd("p", role="alpha")
|
|
76
|
+
cmd_b = runner.build_agent_cmd("p", role="beta")
|
|
77
|
+
self.assertIn("--model", cmd_a)
|
|
78
|
+
self.assertEqual(cmd_a[cmd_a.index("--model") + 1], "env-model")
|
|
79
|
+
self.assertEqual(cmd_b[:2], ["gemini", "-p"])
|
|
80
|
+
|
|
81
|
+
def test_cli_overrides_beat_config(self):
|
|
82
|
+
with mock.patch.dict(os.environ, {"IUMBTEMS_MODEL_ALPHA": "env-model"}, clear=False):
|
|
83
|
+
runner = SwarmRunner(
|
|
84
|
+
mock_mode=True,
|
|
85
|
+
base_dir=Path("/tmp/no-such-research-dir"),
|
|
86
|
+
agent_overrides={"alpha": {"model": "cli-model"}},
|
|
87
|
+
)
|
|
88
|
+
cmd = runner.build_agent_cmd("p", role="alpha")
|
|
89
|
+
self.assertEqual(cmd[cmd.index("--model") + 1], "cli-model")
|
|
90
|
+
|
|
91
|
+
def test_mock_mode_never_spawns(self):
|
|
92
|
+
"""mock_mode short-circuits before cmd construction."""
|
|
93
|
+
runner = SwarmRunner(
|
|
94
|
+
mock_mode=True,
|
|
95
|
+
base_dir=Path("/tmp/no-such-research-dir"),
|
|
96
|
+
agent_overrides={"alpha": {"model": "should-not-be-used"}},
|
|
97
|
+
)
|
|
98
|
+
with mock.patch("subprocess.run") as spawn:
|
|
99
|
+
out = runner.run_claude_process("Orchestrator: build manifest.json")
|
|
100
|
+
spawn.assert_not_called()
|
|
101
|
+
# Canned orchestrator response is scope-decomposition JSON.
|
|
102
|
+
self.assertIn("scopes", out)
|
|
103
|
+
|
|
104
|
+
def test_system_prompt_appended_when_present(self):
|
|
105
|
+
runner = SwarmRunner(mock_mode=True, base_dir=Path("/tmp/no-such-research-dir"))
|
|
106
|
+
existing = PROJECT_ROOT / "prompts" / "agent_alpha_thesis.md"
|
|
107
|
+
cmd = runner.build_agent_cmd("p", system_prompt_file=existing, role="alpha")
|
|
108
|
+
self.assertIn("--system-prompt", cmd)
|
|
109
|
+
self.assertEqual(cmd[cmd.index("--system-prompt") + 1], str(existing))
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
if __name__ == "__main__":
|
|
113
|
+
unittest.main()
|
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Living Dossiers probe test (Stream C falsification criterion).
|
|
3
|
+
|
|
4
|
+
[HYPOTHESIS: injecting a simulated retraction into 5 completed dossiers
|
|
5
|
+
marks >= 4 dependent claim sets STALE within one auditor pass]
|
|
6
|
+
|
|
7
|
+
This test IS the probe: 5 scope dossiers cite shared source hashes; a
|
|
8
|
+
simulated retraction event is injected; exactly one degradation pass runs;
|
|
9
|
+
>= 4 claim sets must end STALE (or SUSPECT for REVISED events).
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
import sys
|
|
14
|
+
import tempfile
|
|
15
|
+
import unittest
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
|
|
19
|
+
if str(PROJECT_ROOT) not in sys.path:
|
|
20
|
+
sys.path.insert(0, str(PROJECT_ROOT))
|
|
21
|
+
|
|
22
|
+
from runner.claim_witness import ( # noqa: E402
|
|
23
|
+
STATUS_LIVE,
|
|
24
|
+
STATUS_STALE,
|
|
25
|
+
STATUS_SUSPECT,
|
|
26
|
+
)
|
|
27
|
+
from runner.living_dossiers import ( # noqa: E402
|
|
28
|
+
EVENT_RETRACTED,
|
|
29
|
+
EVENT_REVISED,
|
|
30
|
+
RetractionEvent,
|
|
31
|
+
apply_degradation,
|
|
32
|
+
check_staleness,
|
|
33
|
+
load_requeue,
|
|
34
|
+
load_retractions,
|
|
35
|
+
write_retraction,
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def build_fixture(base: Path, n_scopes: int = 5) -> dict:
|
|
40
|
+
"""5 completed scopes; scopes 0-3 cite SHARED_A, scope 4 cites SHARED_B."""
|
|
41
|
+
hashes = {}
|
|
42
|
+
sources = base / "sources"
|
|
43
|
+
sources.mkdir(parents=True, exist_ok=True)
|
|
44
|
+
for name, content in (
|
|
45
|
+
("SHARED_A", "Poseidon prover executes the round constraints in 184ms."),
|
|
46
|
+
("SHARED_B", "Independent replication confirmed 150k ops/sec throughput."),
|
|
47
|
+
):
|
|
48
|
+
(sources / f"{name}.json").write_text(
|
|
49
|
+
json.dumps({"hash": name, "url": f"https://x/{name}", "title": name})
|
|
50
|
+
)
|
|
51
|
+
(sources / f"{name}.md").write_text(content)
|
|
52
|
+
hashes[name] = content
|
|
53
|
+
|
|
54
|
+
scratch = base / "scratchpads"
|
|
55
|
+
for i in range(n_scopes):
|
|
56
|
+
scope = scratch / f"scope_{i:02d}"
|
|
57
|
+
scope.mkdir(parents=True, exist_ok=True)
|
|
58
|
+
src = "SHARED_A" if i < 4 else "SHARED_B"
|
|
59
|
+
quote = "184ms" if src == "SHARED_A" else "150k ops/sec"
|
|
60
|
+
(scope / "alpha_dossier.json").write_text(
|
|
61
|
+
json.dumps(
|
|
62
|
+
{
|
|
63
|
+
"agent": "Agent Alpha",
|
|
64
|
+
"scope_id": scope.name,
|
|
65
|
+
"affirmative_claims": [
|
|
66
|
+
{
|
|
67
|
+
"claim_id": f"A-{i:02d}",
|
|
68
|
+
"tag": "VERIFIED",
|
|
69
|
+
"statement": f"scope {i} affirmative",
|
|
70
|
+
"source_hash": src,
|
|
71
|
+
"verbatim_quote": quote,
|
|
72
|
+
}
|
|
73
|
+
],
|
|
74
|
+
"inferred_implications": [],
|
|
75
|
+
"negative_knowledge": [],
|
|
76
|
+
}
|
|
77
|
+
)
|
|
78
|
+
)
|
|
79
|
+
(scope / "beta_dossier.json").write_text(
|
|
80
|
+
json.dumps(
|
|
81
|
+
{
|
|
82
|
+
"agent": "Agent Beta",
|
|
83
|
+
"scope_id": scope.name,
|
|
84
|
+
"falsification_claims": [
|
|
85
|
+
{
|
|
86
|
+
"claim_id": f"B-{i:02d}",
|
|
87
|
+
"tag": "VERIFIED",
|
|
88
|
+
"statement": f"scope {i} falsification",
|
|
89
|
+
"source_hash": src,
|
|
90
|
+
"verbatim_quote": quote,
|
|
91
|
+
"severity": "LOW",
|
|
92
|
+
}
|
|
93
|
+
],
|
|
94
|
+
"methodological_critiques": [],
|
|
95
|
+
"negative_knowledge": [],
|
|
96
|
+
}
|
|
97
|
+
)
|
|
98
|
+
)
|
|
99
|
+
return hashes
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
class TestLivingDossiersProbe(unittest.TestCase):
|
|
103
|
+
def test_probe_retraction_marks_four_claim_sets_stale(self):
|
|
104
|
+
"""THE probe: 5 dossiers, simulated retraction, >=4 claim sets STALE."""
|
|
105
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
106
|
+
base = Path(tmp)
|
|
107
|
+
build_fixture(base, n_scopes=5)
|
|
108
|
+
|
|
109
|
+
# Simulate retraction of SHARED_A (cited by scopes 0-3 = 4 sets).
|
|
110
|
+
write_retraction(
|
|
111
|
+
base,
|
|
112
|
+
"SHARED_A",
|
|
113
|
+
EVENT_RETRACTED,
|
|
114
|
+
note="simulated retraction: paper withdrawn by authors",
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
# Exactly one auditor/degradation pass.
|
|
118
|
+
summary = check_staleness(base)
|
|
119
|
+
|
|
120
|
+
degraded = set(summary["degraded_scopes"])
|
|
121
|
+
# >= 4 claim sets (scopes) must degrade to STALE.
|
|
122
|
+
self.assertGreaterEqual(
|
|
123
|
+
len(degraded), 4, f"expected >=4 degraded scopes, got {degraded}"
|
|
124
|
+
)
|
|
125
|
+
self.assertEqual(degraded, {"scope_00", "scope_01", "scope_02", "scope_03"})
|
|
126
|
+
|
|
127
|
+
# Scope 04 (SHARED_B) must remain LIVE.
|
|
128
|
+
self.assertNotIn("scope_04", degraded)
|
|
129
|
+
|
|
130
|
+
# Ledger is append-only and records the transitions.
|
|
131
|
+
ledger_path = base / "ledger" / "claim_status.json"
|
|
132
|
+
self.assertTrue(ledger_path.exists())
|
|
133
|
+
ledger = json.loads(ledger_path.read_text())
|
|
134
|
+
self.assertGreaterEqual(len(ledger), 8) # 4 scopes x (alpha+beta claims)
|
|
135
|
+
self.assertTrue(all(e["to_status"] == STATUS_STALE for e in ledger))
|
|
136
|
+
|
|
137
|
+
# Requeue sidecar lists exactly the degraded scopes.
|
|
138
|
+
requeue = load_requeue(base)
|
|
139
|
+
self.assertEqual(
|
|
140
|
+
sorted(e["scope_id"] for e in requeue),
|
|
141
|
+
["scope_00", "scope_01", "scope_02", "scope_03"],
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
# Dossiers themselves are NEVER mutated (historical record).
|
|
145
|
+
d = json.loads((base / "scratchpads" / "scope_00" / "alpha_dossier.json").read_text())
|
|
146
|
+
self.assertNotIn("status", d["affirmative_claims"][0])
|
|
147
|
+
self.assertEqual(d["affirmative_claims"][0]["claim_id"], "A-00")
|
|
148
|
+
|
|
149
|
+
def test_revised_marks_suspect_not_stale(self):
|
|
150
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
151
|
+
base = Path(tmp)
|
|
152
|
+
build_fixture(base, n_scopes=2)
|
|
153
|
+
write_retraction(base, "SHARED_A", EVENT_REVISED, note="benchmark table updated")
|
|
154
|
+
summary = check_staleness(base)
|
|
155
|
+
self.assertIn("scope_00", summary["degraded_scopes"])
|
|
156
|
+
|
|
157
|
+
# apply_degradation unit-level: SUSPECT for REVISED
|
|
158
|
+
from runner.claim_witness import normalize_claim
|
|
159
|
+
|
|
160
|
+
claim = normalize_claim(
|
|
161
|
+
{"claim_id": "C1", "tag": "VERIFIED", "source_hash": "SHARED_A", "verbatim_quote": "184ms"}
|
|
162
|
+
)
|
|
163
|
+
retr = load_retractions(base)
|
|
164
|
+
degraded, events = apply_degradation([claim], retr, scope_id="s")
|
|
165
|
+
self.assertEqual(degraded[0].status, STATUS_SUSPECT)
|
|
166
|
+
self.assertEqual(events[0].to_status, STATUS_SUSPECT)
|
|
167
|
+
|
|
168
|
+
def test_no_retractions_no_degradation(self):
|
|
169
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
170
|
+
base = Path(tmp)
|
|
171
|
+
build_fixture(base, n_scopes=5)
|
|
172
|
+
summary = check_staleness(base)
|
|
173
|
+
self.assertEqual(summary["retractions"], 0)
|
|
174
|
+
self.assertEqual(summary["degraded_scopes"], [])
|
|
175
|
+
self.assertEqual(summary["status_events"], 0)
|
|
176
|
+
|
|
177
|
+
def test_retraction_event_validation(self):
|
|
178
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
179
|
+
base = Path(tmp)
|
|
180
|
+
with self.assertRaises(ValueError):
|
|
181
|
+
write_retraction(base, "H", "EXPLODED", note="bad event")
|
|
182
|
+
|
|
183
|
+
def test_apply_degradation_idempotent(self):
|
|
184
|
+
"""Second pass over already-STALE claims produces no new events."""
|
|
185
|
+
from runner.claim_witness import normalize_claim
|
|
186
|
+
|
|
187
|
+
retr = {
|
|
188
|
+
"SHARED_A": RetractionEvent(
|
|
189
|
+
hash="SHARED_A", event=EVENT_RETRACTED, at="t", note="n"
|
|
190
|
+
)
|
|
191
|
+
}
|
|
192
|
+
claim = normalize_claim(
|
|
193
|
+
{"claim_id": "C1", "tag": "VERIFIED", "source_hash": "SHARED_A", "verbatim_quote": "q"}
|
|
194
|
+
)
|
|
195
|
+
d1, ev1 = apply_degradation([claim], retr, scope_id="s")
|
|
196
|
+
self.assertEqual(len(ev1), 1)
|
|
197
|
+
self.assertEqual(d1[0].status, STATUS_STALE)
|
|
198
|
+
d2, ev2 = apply_degradation(d1, retr, scope_id="s")
|
|
199
|
+
self.assertEqual(len(ev2), 0, "already-degraded claims produce no new events")
|
|
200
|
+
self.assertEqual(d2[0].status, STATUS_STALE)
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
if __name__ == "__main__":
|
|
204
|
+
unittest.main()
|