@heretek-ai/epistemic-swarm 0.2.2 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (138) hide show
  1. package/.agents/skills/brainstorming/SKILL.md +13 -0
  2. package/.agents/skills/code_audit/SKILL.md +13 -0
  3. package/.agents/skills/epistemic_search/SKILL.md +13 -0
  4. package/.agents/skills/grilling/SKILL.md +13 -0
  5. package/.agents/skills/oss_scout/SKILL.md +13 -0
  6. package/.agents/skills/research_cache/SKILL.md +13 -0
  7. package/.agents/skills/swarm_config/SKILL.md +13 -0
  8. package/.claude-plugin/plugin.json +15 -5
  9. package/.omp/README.md +39 -0
  10. package/.omp/SYSTEM.md +12 -0
  11. package/.omp/commands/audit.md +10 -0
  12. package/.omp/commands/brainstorming.md +12 -0
  13. package/.omp/commands/grill.md +9 -0
  14. package/.omp/commands/scout.md +10 -0
  15. package/.omp/commands/swarm-config.md +10 -0
  16. package/.omp/commands/swarm.md +9 -0
  17. package/.omp/hooks/post/epistemic-audit.ts +16 -0
  18. package/.omp/hooks/pre/epistemic-redirect.ts +21 -0
  19. package/.omp/prompts/brainstorming.md +8 -0
  20. package/.omp/prompts/swarm.md +8 -0
  21. package/MARKETPLACE.md +8 -0
  22. package/README.md +53 -8
  23. package/bin/cli.js +27 -3
  24. package/config/domain_packs/biopharma.json +23 -0
  25. package/config/domain_packs/legal.json +19 -0
  26. package/config/domain_packs/quant.json +19 -0
  27. package/config/mcp-research-servers.json +7 -0
  28. package/config/mcp_launcher.py +48 -137
  29. package/config/opencode-snippet.json +58 -3
  30. package/config/searxng_mcp.py +42 -83
  31. package/extensions/pi/index.js +196 -28
  32. package/install.sh +20 -4
  33. package/package.json +40 -5
  34. package/plugins/antigravity/README.md +28 -0
  35. package/plugins/antigravity/agents/alpha-thesis.md +6 -0
  36. package/plugins/antigravity/agents/beta-antithesis.md +7 -0
  37. package/plugins/antigravity/agents/brainstormer.md +7 -0
  38. package/plugins/antigravity/agents/epistemic-auditor.md +5 -0
  39. package/plugins/antigravity/hooks.json +23 -0
  40. package/plugins/antigravity/mcp_config.json +33 -0
  41. package/plugins/antigravity/plugin.json +21 -0
  42. package/plugins/antigravity/rules/epistemic-integrity.md +6 -0
  43. package/plugins/antigravity/skills/brainstorming/SKILL.md +13 -0
  44. package/plugins/antigravity/skills/code_audit/SKILL.md +13 -0
  45. package/plugins/antigravity/skills/epistemic_search/SKILL.md +13 -0
  46. package/plugins/antigravity/skills/grilling/SKILL.md +13 -0
  47. package/plugins/antigravity/skills/oss_scout/SKILL.md +13 -0
  48. package/plugins/antigravity/skills/research_cache/SKILL.md +13 -0
  49. package/plugins/antigravity/skills/swarm_config/SKILL.md +13 -0
  50. package/plugins/codex/AGENTS.md.snippet +10 -0
  51. package/plugins/codex/README.md +37 -0
  52. package/plugins/codex/config.toml.snippet +28 -0
  53. package/plugins/codex/openai.yaml +24 -0
  54. package/plugins/codex/skills/brainstorming/SKILL.md +13 -0
  55. package/plugins/codex/skills/code_audit/SKILL.md +13 -0
  56. package/plugins/codex/skills/epistemic_search/SKILL.md +13 -0
  57. package/plugins/codex/skills/grilling/SKILL.md +13 -0
  58. package/plugins/codex/skills/oss_scout/SKILL.md +13 -0
  59. package/plugins/codex/skills/research_cache/SKILL.md +13 -0
  60. package/plugins/codex/skills/swarm_config/SKILL.md +13 -0
  61. package/plugins/gemini/GEMINI.md +15 -0
  62. package/plugins/gemini/README.md +19 -0
  63. package/plugins/gemini/commands/audit.toml +6 -0
  64. package/plugins/gemini/commands/brainstorming.toml +10 -0
  65. package/plugins/gemini/commands/grill.toml +6 -0
  66. package/plugins/gemini/commands/scout.toml +7 -0
  67. package/plugins/gemini/commands/swarm-config.toml +7 -0
  68. package/plugins/gemini/commands/swarm.toml +8 -0
  69. package/plugins/gemini/gemini-extension.json +38 -0
  70. package/plugins/gemini/hooks/hooks.json +11 -0
  71. package/plugins/gemini/skills/brainstorming/SKILL.md +13 -0
  72. package/plugins/gemini/skills/code_audit/SKILL.md +13 -0
  73. package/plugins/gemini/skills/epistemic_search/SKILL.md +13 -0
  74. package/plugins/gemini/skills/grilling/SKILL.md +13 -0
  75. package/plugins/gemini/skills/oss_scout/SKILL.md +13 -0
  76. package/plugins/gemini/skills/research_cache/SKILL.md +13 -0
  77. package/plugins/gemini/skills/swarm_config/SKILL.md +13 -0
  78. package/plugins/opencode/index.js +335 -118
  79. package/prompts/agent_brainstormer.md +97 -0
  80. package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
  81. package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
  82. package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
  83. package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
  84. package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
  85. package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
  86. package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
  87. package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
  88. package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
  89. package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
  90. package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
  91. package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
  92. package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
  93. package/runner/auctioneer.py +169 -0
  94. package/runner/auditor_engine.py +127 -12
  95. package/runner/claim_store.py +361 -0
  96. package/runner/claim_witness.py +183 -0
  97. package/runner/living_dossiers.py +355 -0
  98. package/runner/mcp_protocol.py +188 -0
  99. package/runner/mcp_server.py +567 -0
  100. package/runner/pcrb.py +212 -0
  101. package/runner/pcrb_verify.py +237 -0
  102. package/runner/refinement.py +335 -0
  103. package/runner/research_swarm.py +378 -94
  104. package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
  105. package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
  106. package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
  107. package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
  108. package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
  109. package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
  110. package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
  111. package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
  112. package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
  113. package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
  114. package/runner/tests/fixtures/auction_objective.json +35 -0
  115. package/runner/tests/fixtures/borderline_claims.json +27 -0
  116. package/runner/tests/fixtures/divergence_objectives.json +31 -0
  117. package/runner/tests/test_auction_order.py +173 -0
  118. package/runner/tests/test_claim_store.py +162 -0
  119. package/runner/tests/test_claim_witness.py +186 -0
  120. package/runner/tests/test_domain_packs.py +217 -0
  121. package/runner/tests/test_fleet_seam.py +113 -0
  122. package/runner/tests/test_living_dossiers.py +204 -0
  123. package/runner/tests/test_mcp_server.py +212 -0
  124. package/runner/tests/test_pcrb.py +240 -0
  125. package/runner/tests/test_refinement.py +255 -0
  126. package/runner/tests/test_swarm.py +173 -15
  127. package/scripts/auction_experiment.py +180 -0
  128. package/scripts/build_adapters.py +183 -0
  129. package/scripts/divergence_experiment.py +184 -0
  130. package/skills/brainstorming/SKILL.md +106 -0
  131. package/skills/brainstorming/__init__.py +1 -0
  132. package/skills/brainstorming/scripts/brainstorm.py +200 -0
  133. package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
  134. package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
  135. package/skills/swarm_config/SKILL.md +1 -1
  136. package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
  137. package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
  138. package/skills/swarm_config/configure.py +45 -11
@@ -0,0 +1,217 @@
1
+ #!/usr/bin/env python3
2
+ """Regulated Domain Packs probe test (Stream G falsification criterion).
3
+
4
+ [HYPOTHESIS: a biopharma constitution flips accept/reject on >= 25% of a
5
+ 20-claim borderline fixture set]
6
+
7
+ This test IS the probe: the 20-claim borderline fixture is judged under the
8
+ legacy default constitution and under the biopharma pack; >= 25% of per-claim
9
+ verdicts must flip. The zero-tolerance retraction policy must downgrade
10
+ claims with retracted sources regardless of score.
11
+ """
12
+
13
+ import json
14
+ import sys
15
+ import unittest
16
+ from pathlib import Path
17
+
18
+ PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
19
+ if str(PROJECT_ROOT) not in sys.path:
20
+ sys.path.insert(0, str(PROJECT_ROOT))
21
+
22
+ from runner.claim_witness import normalize_claim # noqa: E402
23
+ from runner.refinement import ( # noqa: E402
24
+ LEGACY_CONSTITUTION,
25
+ claim_verdict,
26
+ compute_epistemic_score_from_claims,
27
+ load_domain_pack,
28
+ )
29
+
30
+ FIXTURE = PROJECT_ROOT / "runner" / "tests" / "fixtures" / "borderline_claims.json"
31
+ N_CLAIMS = 20
32
+ FLIP_BAR = 0.25
33
+
34
+
35
+ def load_fixture():
36
+ data = json.loads(FIXTURE.read_text(encoding="utf-8"))
37
+ claims = [normalize_claim(c) for c in data["claims"]]
38
+ return claims, set(data["retracted_hashes"])
39
+
40
+
41
+ class TestDomainPacksProbe(unittest.TestCase):
42
+ @classmethod
43
+ def setUpClass(cls):
44
+ cls.claims, cls.retracted = load_fixture()
45
+ cls.biopharma = load_domain_pack("biopharma")
46
+
47
+ def test_fixture_shape(self):
48
+ self.assertEqual(len(self.claims), N_CLAIMS)
49
+
50
+ def test_load_all_three_packs(self):
51
+ for pack_id in ("biopharma", "quant", "legal"):
52
+ c = load_domain_pack(pack_id)
53
+ self.assertIsNotNone(c.tier_weights)
54
+ self.assertIn("PREPRINT", c.tier_weights)
55
+ with self.assertRaises(FileNotFoundError):
56
+ load_domain_pack("no-such-pack")
57
+
58
+ def test_probe_biopharma_flips_at_least_quarter(self):
59
+ """THE probe: >= 25% of 20 borderline verdicts flip under biopharma."""
60
+ default_results = [
61
+ claim_verdict(c, constitution=LEGACY_CONSTITUTION, retracted_hashes=self.retracted)
62
+ for c in self.claims
63
+ ]
64
+ bio_results = [
65
+ claim_verdict(c, constitution=self.biopharma, retracted_hashes=self.retracted)
66
+ for c in self.claims
67
+ ]
68
+
69
+ flips = [
70
+ (d["claim_id"], d["verdict"], b["verdict"])
71
+ for d, b in zip(default_results, bio_results)
72
+ if d["verdict"] != b["verdict"]
73
+ ]
74
+ flip_rate = len(flips) / N_CLAIMS
75
+ self.assertGreaterEqual(
76
+ flip_rate,
77
+ FLIP_BAR,
78
+ f"flip rate {flip_rate:.0%} ({len(flips)}/{N_CLAIMS}) below {FLIP_BAR:.0%}: {flips}",
79
+ )
80
+
81
+ # Default constitution accepts the whole borderline set...
82
+ self.assertTrue(all(r["verdict"] == "ACCEPTED" for r in default_results),
83
+ [r for r in default_results if r["verdict"] != "ACCEPTED"])
84
+ # ...and every flip is default-ACCEPTED -> biopharma-REJECTED.
85
+ self.assertTrue(all(d == "ACCEPTED" and b == "REJECTED" for _c, d, b in flips))
86
+
87
+ def test_banned_domain_mechanism(self):
88
+ banned = [c for c in self.claims if "mirror.example" in (c.source_url or "") or "predatory-journal" in (c.source_url or "")]
89
+ self.assertEqual(len(banned), 6, "fixture expects 6 banned-domain claims")
90
+ for c in banned:
91
+ r = claim_verdict(c, constitution=self.biopharma)
92
+ self.assertEqual(r["verdict"], "REJECTED", c.claim_id)
93
+ self.assertTrue(any("BANNED_DOMAIN" in x for x in r["reasons"]), r["reasons"])
94
+
95
+ def test_missing_tag_mechanism(self):
96
+ untagged = [c for c in self.claims if not c.tag]
97
+ self.assertEqual(len(untagged), 2)
98
+ for c in untagged:
99
+ r = claim_verdict(c, constitution=self.biopharma)
100
+ self.assertEqual(r["verdict"], "REJECTED", c.claim_id)
101
+ self.assertTrue(any("MISSING_TAG" in x for x in r["reasons"]), r["reasons"])
102
+
103
+ def test_zero_tolerance_retraction_mechanism(self):
104
+ """RETRACTED source downgrades regardless of quote match or score."""
105
+ retracted_claims = [c for c in self.claims if c.source_hash in self.retracted]
106
+ self.assertEqual(len(retracted_claims), 2)
107
+ for c in retracted_claims:
108
+ # Even without passing retracted_hashes: status-based downgrade.
109
+ c.status = "STALE"
110
+ r = claim_verdict(c, constitution=self.biopharma)
111
+ self.assertEqual(r["verdict"], "REJECTED", c.claim_id)
112
+ self.assertTrue(
113
+ any("ZERO_TOLERANCE_RETRACTION" in x for x in r["reasons"]), r["reasons"]
114
+ )
115
+ # Standard policy does NOT apply this rule.
116
+ r_std = claim_verdict(c, constitution=LEGACY_CONSTITUTION)
117
+ self.assertEqual(r_std["verdict"], "ACCEPTED")
118
+
119
+ def test_tier_weights_move_the_score(self):
120
+ """Preprint downweighting lowers E(D); threshold also comes from the pack."""
121
+ s_default, _ = compute_epistemic_score_from_claims(self.claims, constitution=LEGACY_CONSTITUTION)
122
+ s_bio, br = compute_epistemic_score_from_claims(self.claims, constitution=self.biopharma)
123
+ self.assertLess(s_bio, s_default, (s_default, s_bio))
124
+ # 18 of 20 claims carry the VERIFIED tag (2 are deliberately untagged);
125
+ # each scores 0.3 under the biopharma preprint weight.
126
+ n_verified = sum(1 for c in self.claims if c.tag == "VERIFIED")
127
+ self.assertEqual(n_verified, 18)
128
+ self.assertAlmostEqual(br["verified_weight"], n_verified * 0.3, places=3)
129
+
130
+ def test_legacy_default_preserves_existing_behavior(self):
131
+ """No pack => flat weights, no domain/tag rules, standard retraction."""
132
+ self.assertEqual(LEGACY_CONSTITUTION.weight_for("PREPRINT"), 1.0)
133
+ self.assertEqual(LEGACY_CONSTITUTION.banned_domains, [])
134
+ self.assertEqual(LEGACY_CONSTITUTION.mandatory_tags, [])
135
+ self.assertEqual(LEGACY_CONSTITUTION.retraction_policy, "standard")
136
+ self.assertEqual(LEGACY_CONSTITUTION.accept_threshold, 0.65)
137
+
138
+
139
+ class TestAuditConstitutionWiring(unittest.TestCase):
140
+ def test_audit_scope_default_unchanged(self):
141
+ """audit_scope() with no constitution must match legacy numbers exactly."""
142
+ import tempfile
143
+ from runner.auditor_engine import EpistemicAuditorEngine
144
+ from runner.state_machine import ResearchStateMachine
145
+ from skills.research_cache.hasher import SourceHasher
146
+
147
+ with tempfile.TemporaryDirectory() as tmp:
148
+ base = Path(tmp)
149
+ hasher = SourceHasher(base)
150
+ sm = ResearchStateMachine(base_dir=base)
151
+ shash = hasher.store_source(
152
+ "https://x", "FPGA prover executes Poseidon in 184ms.", "T"
153
+ )
154
+ sm.init_session("t")
155
+ sm.set_scopes([{"scope_id": "s1", "dependencies": []}])
156
+ sm.record_agent_completion("s1", "alpha", {
157
+ "affirmative_claims": [{
158
+ "claim_id": "A1", "tag": "VERIFIED", "statement": "s",
159
+ "source_hash": shash, "verbatim_quote": "executes Poseidon in 184ms",
160
+ }],
161
+ "inferred_implications": [],
162
+ "negative_knowledge": [],
163
+ })
164
+ sm.record_agent_completion("s1", "beta", {
165
+ "falsification_claims": [],
166
+ "methodological_critiques": [],
167
+ "negative_knowledge": [],
168
+ })
169
+
170
+ engine = EpistemicAuditorEngine(base_dir=base)
171
+ report = engine.audit_scope("s1")
172
+ self.assertEqual(report["summary"]["verified_passed"], 1)
173
+ self.assertEqual(report["summary"]["verdict"], "CERTIFIED")
174
+
175
+ def test_audit_scope_with_biopharma_uses_threshold(self):
176
+ import tempfile
177
+ from runner.auditor_engine import EpistemicAuditorEngine
178
+ from runner.refinement import load_domain_pack
179
+ from runner.state_machine import ResearchStateMachine
180
+ from skills.research_cache.hasher import SourceHasher
181
+
182
+ with tempfile.TemporaryDirectory() as tmp:
183
+ base = Path(tmp)
184
+ hasher = SourceHasher(base)
185
+ sm = ResearchStateMachine(base_dir=base)
186
+ shash = hasher.store_source(
187
+ "https://predatory-journal.example/x",
188
+ "Weak preprint finding about surrogate endpoints.",
189
+ "T", tier="PREPRINT",
190
+ )
191
+ sm.init_session("t")
192
+ sm.set_scopes([{"scope_id": "s1", "dependencies": []}])
193
+ sm.record_agent_completion("s1", "alpha", {
194
+ "affirmative_claims": [{
195
+ "claim_id": "A1", "tag": "VERIFIED", "statement": "s",
196
+ "source_hash": shash,
197
+ "source_url": "https://predatory-journal.example/x",
198
+ "verbatim_quote": "surrogate endpoints",
199
+ }],
200
+ "inferred_implications": [],
201
+ "negative_knowledge": [],
202
+ })
203
+ sm.record_agent_completion("s1", "beta", {
204
+ "falsification_claims": [],
205
+ "methodological_critiques": [],
206
+ "negative_knowledge": [],
207
+ })
208
+
209
+ engine = EpistemicAuditorEngine(base_dir=base)
210
+ report = engine.audit_scope("s1", constitution=load_domain_pack("biopharma"))
211
+ # Banned domain claim => REJECTED under biopharma.
212
+ self.assertEqual(report["summary"]["unverified_rejected"], 1)
213
+ self.assertEqual(report["summary"]["verified_passed"], 0)
214
+
215
+
216
+ if __name__ == "__main__":
217
+ unittest.main()
@@ -0,0 +1,113 @@
1
+ #!/usr/bin/env python3
2
+ """Tests for the Stream E model-diverse adversary fleet seam.
3
+
4
+ The seam is config-first (per-agent backend/model), CLI-overridable, with env
5
+ fallback. Mock mode must NEVER build or spawn these commands.
6
+ """
7
+
8
+ import os
9
+ import sys
10
+ import unittest
11
+ from pathlib import Path
12
+ from unittest import mock
13
+
14
+ PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
15
+ if str(PROJECT_ROOT) not in sys.path:
16
+ sys.path.insert(0, str(PROJECT_ROOT))
17
+
18
+ from runner.research_swarm import SwarmRunner # noqa: E402
19
+
20
+
21
+ class TestFleetSeam(unittest.TestCase):
22
+ def test_default_backend_is_claude_with_no_model(self):
23
+ runner = SwarmRunner(mock_mode=True, base_dir=Path("/tmp/no-such-research-dir"))
24
+ cmd = runner.build_agent_cmd("prompt", role="alpha")
25
+ self.assertEqual(cmd[0:2], ["claude", "-p"])
26
+ self.assertNotIn("--model", cmd)
27
+ self.assertIn("prompt", cmd)
28
+
29
+ def test_model_config_adds_model_flag(self):
30
+ runner = SwarmRunner(
31
+ mock_mode=True,
32
+ base_dir=Path("/tmp/no-such-research-dir"),
33
+ agent_overrides={"alpha": {"model": "claude-opus-5"}},
34
+ )
35
+ cmd = runner.build_agent_cmd("prompt", role="alpha")
36
+ self.assertIn("--model", cmd)
37
+ self.assertEqual(cmd[cmd.index("--model") + 1], "claude-opus-5")
38
+
39
+ def test_beta_backend_command_list(self):
40
+ """Cross-FAMILY means non-Claude processes; backend is a cmd list."""
41
+ runner = SwarmRunner(
42
+ mock_mode=True,
43
+ base_dir=Path("/tmp/no-such-research-dir"),
44
+ agent_overrides={"beta": {"backend": ["ollama", "run", "qwen3"], "model": None}},
45
+ )
46
+ cmd = runner.build_agent_cmd("prompt", role="beta")
47
+ self.assertEqual(cmd[:3], ["ollama", "run", "qwen3"])
48
+ self.assertNotIn("--model", cmd)
49
+
50
+ def test_roles_resolve_independently(self):
51
+ """Asymmetry is the point: alpha and beta may use different families."""
52
+ runner = SwarmRunner(
53
+ mock_mode=True,
54
+ base_dir=Path("/tmp/no-such-research-dir"),
55
+ agent_overrides={
56
+ "alpha": {"model": "claude-opus-5"},
57
+ "beta": {"backend": ["codex", "exec"], "model": "gpt-5"},
58
+ },
59
+ )
60
+ cmd_a = runner.build_agent_cmd("p", role="alpha")
61
+ cmd_b = runner.build_agent_cmd("p", role="beta")
62
+ self.assertEqual(cmd_a[0:2], ["claude", "-p"])
63
+ self.assertIn("--model", cmd_a)
64
+ self.assertEqual(cmd_b[:2], ["codex", "exec"])
65
+ self.assertIn("--model", cmd_b)
66
+ self.assertEqual(cmd_b[cmd_b.index("--model") + 1], "gpt-5")
67
+
68
+ def test_env_fallback(self):
69
+ with mock.patch.dict(
70
+ os.environ,
71
+ {"IUMBTEMS_MODEL_ALPHA": "env-model", "IUMBTEMS_BACKEND_BETA": "gemini -p"},
72
+ clear=False,
73
+ ):
74
+ runner = SwarmRunner(mock_mode=True, base_dir=Path("/tmp/no-such-research-dir"))
75
+ cmd_a = runner.build_agent_cmd("p", role="alpha")
76
+ cmd_b = runner.build_agent_cmd("p", role="beta")
77
+ self.assertIn("--model", cmd_a)
78
+ self.assertEqual(cmd_a[cmd_a.index("--model") + 1], "env-model")
79
+ self.assertEqual(cmd_b[:2], ["gemini", "-p"])
80
+
81
+ def test_cli_overrides_beat_config(self):
82
+ with mock.patch.dict(os.environ, {"IUMBTEMS_MODEL_ALPHA": "env-model"}, clear=False):
83
+ runner = SwarmRunner(
84
+ mock_mode=True,
85
+ base_dir=Path("/tmp/no-such-research-dir"),
86
+ agent_overrides={"alpha": {"model": "cli-model"}},
87
+ )
88
+ cmd = runner.build_agent_cmd("p", role="alpha")
89
+ self.assertEqual(cmd[cmd.index("--model") + 1], "cli-model")
90
+
91
+ def test_mock_mode_never_spawns(self):
92
+ """mock_mode short-circuits before cmd construction."""
93
+ runner = SwarmRunner(
94
+ mock_mode=True,
95
+ base_dir=Path("/tmp/no-such-research-dir"),
96
+ agent_overrides={"alpha": {"model": "should-not-be-used"}},
97
+ )
98
+ with mock.patch("subprocess.run") as spawn:
99
+ out = runner.run_claude_process("Orchestrator: build manifest.json")
100
+ spawn.assert_not_called()
101
+ # Canned orchestrator response is scope-decomposition JSON.
102
+ self.assertIn("scopes", out)
103
+
104
+ def test_system_prompt_appended_when_present(self):
105
+ runner = SwarmRunner(mock_mode=True, base_dir=Path("/tmp/no-such-research-dir"))
106
+ existing = PROJECT_ROOT / "prompts" / "agent_alpha_thesis.md"
107
+ cmd = runner.build_agent_cmd("p", system_prompt_file=existing, role="alpha")
108
+ self.assertIn("--system-prompt", cmd)
109
+ self.assertEqual(cmd[cmd.index("--system-prompt") + 1], str(existing))
110
+
111
+
112
+ if __name__ == "__main__":
113
+ unittest.main()
@@ -0,0 +1,204 @@
1
+ #!/usr/bin/env python3
2
+ """Living Dossiers probe test (Stream C falsification criterion).
3
+
4
+ [HYPOTHESIS: injecting a simulated retraction into 5 completed dossiers
5
+ marks >= 4 dependent claim sets STALE within one auditor pass]
6
+
7
+ This test IS the probe: 5 scope dossiers cite shared source hashes; a
8
+ simulated retraction event is injected; exactly one degradation pass runs;
9
+ >= 4 claim sets must end STALE (or SUSPECT for REVISED events).
10
+ """
11
+
12
+ import json
13
+ import sys
14
+ import tempfile
15
+ import unittest
16
+ from pathlib import Path
17
+
18
+ PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
19
+ if str(PROJECT_ROOT) not in sys.path:
20
+ sys.path.insert(0, str(PROJECT_ROOT))
21
+
22
+ from runner.claim_witness import ( # noqa: E402
23
+ STATUS_LIVE,
24
+ STATUS_STALE,
25
+ STATUS_SUSPECT,
26
+ )
27
+ from runner.living_dossiers import ( # noqa: E402
28
+ EVENT_RETRACTED,
29
+ EVENT_REVISED,
30
+ RetractionEvent,
31
+ apply_degradation,
32
+ check_staleness,
33
+ load_requeue,
34
+ load_retractions,
35
+ write_retraction,
36
+ )
37
+
38
+
39
+ def build_fixture(base: Path, n_scopes: int = 5) -> dict:
40
+ """5 completed scopes; scopes 0-3 cite SHARED_A, scope 4 cites SHARED_B."""
41
+ hashes = {}
42
+ sources = base / "sources"
43
+ sources.mkdir(parents=True, exist_ok=True)
44
+ for name, content in (
45
+ ("SHARED_A", "Poseidon prover executes the round constraints in 184ms."),
46
+ ("SHARED_B", "Independent replication confirmed 150k ops/sec throughput."),
47
+ ):
48
+ (sources / f"{name}.json").write_text(
49
+ json.dumps({"hash": name, "url": f"https://x/{name}", "title": name})
50
+ )
51
+ (sources / f"{name}.md").write_text(content)
52
+ hashes[name] = content
53
+
54
+ scratch = base / "scratchpads"
55
+ for i in range(n_scopes):
56
+ scope = scratch / f"scope_{i:02d}"
57
+ scope.mkdir(parents=True, exist_ok=True)
58
+ src = "SHARED_A" if i < 4 else "SHARED_B"
59
+ quote = "184ms" if src == "SHARED_A" else "150k ops/sec"
60
+ (scope / "alpha_dossier.json").write_text(
61
+ json.dumps(
62
+ {
63
+ "agent": "Agent Alpha",
64
+ "scope_id": scope.name,
65
+ "affirmative_claims": [
66
+ {
67
+ "claim_id": f"A-{i:02d}",
68
+ "tag": "VERIFIED",
69
+ "statement": f"scope {i} affirmative",
70
+ "source_hash": src,
71
+ "verbatim_quote": quote,
72
+ }
73
+ ],
74
+ "inferred_implications": [],
75
+ "negative_knowledge": [],
76
+ }
77
+ )
78
+ )
79
+ (scope / "beta_dossier.json").write_text(
80
+ json.dumps(
81
+ {
82
+ "agent": "Agent Beta",
83
+ "scope_id": scope.name,
84
+ "falsification_claims": [
85
+ {
86
+ "claim_id": f"B-{i:02d}",
87
+ "tag": "VERIFIED",
88
+ "statement": f"scope {i} falsification",
89
+ "source_hash": src,
90
+ "verbatim_quote": quote,
91
+ "severity": "LOW",
92
+ }
93
+ ],
94
+ "methodological_critiques": [],
95
+ "negative_knowledge": [],
96
+ }
97
+ )
98
+ )
99
+ return hashes
100
+
101
+
102
+ class TestLivingDossiersProbe(unittest.TestCase):
103
+ def test_probe_retraction_marks_four_claim_sets_stale(self):
104
+ """THE probe: 5 dossiers, simulated retraction, >=4 claim sets STALE."""
105
+ with tempfile.TemporaryDirectory() as tmp:
106
+ base = Path(tmp)
107
+ build_fixture(base, n_scopes=5)
108
+
109
+ # Simulate retraction of SHARED_A (cited by scopes 0-3 = 4 sets).
110
+ write_retraction(
111
+ base,
112
+ "SHARED_A",
113
+ EVENT_RETRACTED,
114
+ note="simulated retraction: paper withdrawn by authors",
115
+ )
116
+
117
+ # Exactly one auditor/degradation pass.
118
+ summary = check_staleness(base)
119
+
120
+ degraded = set(summary["degraded_scopes"])
121
+ # >= 4 claim sets (scopes) must degrade to STALE.
122
+ self.assertGreaterEqual(
123
+ len(degraded), 4, f"expected >=4 degraded scopes, got {degraded}"
124
+ )
125
+ self.assertEqual(degraded, {"scope_00", "scope_01", "scope_02", "scope_03"})
126
+
127
+ # Scope 04 (SHARED_B) must remain LIVE.
128
+ self.assertNotIn("scope_04", degraded)
129
+
130
+ # Ledger is append-only and records the transitions.
131
+ ledger_path = base / "ledger" / "claim_status.json"
132
+ self.assertTrue(ledger_path.exists())
133
+ ledger = json.loads(ledger_path.read_text())
134
+ self.assertGreaterEqual(len(ledger), 8) # 4 scopes x (alpha+beta claims)
135
+ self.assertTrue(all(e["to_status"] == STATUS_STALE for e in ledger))
136
+
137
+ # Requeue sidecar lists exactly the degraded scopes.
138
+ requeue = load_requeue(base)
139
+ self.assertEqual(
140
+ sorted(e["scope_id"] for e in requeue),
141
+ ["scope_00", "scope_01", "scope_02", "scope_03"],
142
+ )
143
+
144
+ # Dossiers themselves are NEVER mutated (historical record).
145
+ d = json.loads((base / "scratchpads" / "scope_00" / "alpha_dossier.json").read_text())
146
+ self.assertNotIn("status", d["affirmative_claims"][0])
147
+ self.assertEqual(d["affirmative_claims"][0]["claim_id"], "A-00")
148
+
149
+ def test_revised_marks_suspect_not_stale(self):
150
+ with tempfile.TemporaryDirectory() as tmp:
151
+ base = Path(tmp)
152
+ build_fixture(base, n_scopes=2)
153
+ write_retraction(base, "SHARED_A", EVENT_REVISED, note="benchmark table updated")
154
+ summary = check_staleness(base)
155
+ self.assertIn("scope_00", summary["degraded_scopes"])
156
+
157
+ # apply_degradation unit-level: SUSPECT for REVISED
158
+ from runner.claim_witness import normalize_claim
159
+
160
+ claim = normalize_claim(
161
+ {"claim_id": "C1", "tag": "VERIFIED", "source_hash": "SHARED_A", "verbatim_quote": "184ms"}
162
+ )
163
+ retr = load_retractions(base)
164
+ degraded, events = apply_degradation([claim], retr, scope_id="s")
165
+ self.assertEqual(degraded[0].status, STATUS_SUSPECT)
166
+ self.assertEqual(events[0].to_status, STATUS_SUSPECT)
167
+
168
+ def test_no_retractions_no_degradation(self):
169
+ with tempfile.TemporaryDirectory() as tmp:
170
+ base = Path(tmp)
171
+ build_fixture(base, n_scopes=5)
172
+ summary = check_staleness(base)
173
+ self.assertEqual(summary["retractions"], 0)
174
+ self.assertEqual(summary["degraded_scopes"], [])
175
+ self.assertEqual(summary["status_events"], 0)
176
+
177
+ def test_retraction_event_validation(self):
178
+ with tempfile.TemporaryDirectory() as tmp:
179
+ base = Path(tmp)
180
+ with self.assertRaises(ValueError):
181
+ write_retraction(base, "H", "EXPLODED", note="bad event")
182
+
183
+ def test_apply_degradation_idempotent(self):
184
+ """Second pass over already-STALE claims produces no new events."""
185
+ from runner.claim_witness import normalize_claim
186
+
187
+ retr = {
188
+ "SHARED_A": RetractionEvent(
189
+ hash="SHARED_A", event=EVENT_RETRACTED, at="t", note="n"
190
+ )
191
+ }
192
+ claim = normalize_claim(
193
+ {"claim_id": "C1", "tag": "VERIFIED", "source_hash": "SHARED_A", "verbatim_quote": "q"}
194
+ )
195
+ d1, ev1 = apply_degradation([claim], retr, scope_id="s")
196
+ self.assertEqual(len(ev1), 1)
197
+ self.assertEqual(d1[0].status, STATUS_STALE)
198
+ d2, ev2 = apply_degradation(d1, retr, scope_id="s")
199
+ self.assertEqual(len(ev2), 0, "already-degraded claims produce no new events")
200
+ self.assertEqual(d2[0].status, STATUS_STALE)
201
+
202
+
203
+ if __name__ == "__main__":
204
+ unittest.main()