@heretek-ai/epistemic-swarm 0.2.2 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (138) hide show
  1. package/.agents/skills/brainstorming/SKILL.md +13 -0
  2. package/.agents/skills/code_audit/SKILL.md +13 -0
  3. package/.agents/skills/epistemic_search/SKILL.md +13 -0
  4. package/.agents/skills/grilling/SKILL.md +13 -0
  5. package/.agents/skills/oss_scout/SKILL.md +13 -0
  6. package/.agents/skills/research_cache/SKILL.md +13 -0
  7. package/.agents/skills/swarm_config/SKILL.md +13 -0
  8. package/.claude-plugin/plugin.json +15 -5
  9. package/.omp/README.md +39 -0
  10. package/.omp/SYSTEM.md +12 -0
  11. package/.omp/commands/audit.md +10 -0
  12. package/.omp/commands/brainstorming.md +12 -0
  13. package/.omp/commands/grill.md +9 -0
  14. package/.omp/commands/scout.md +10 -0
  15. package/.omp/commands/swarm-config.md +10 -0
  16. package/.omp/commands/swarm.md +9 -0
  17. package/.omp/hooks/post/epistemic-audit.ts +16 -0
  18. package/.omp/hooks/pre/epistemic-redirect.ts +21 -0
  19. package/.omp/prompts/brainstorming.md +8 -0
  20. package/.omp/prompts/swarm.md +8 -0
  21. package/MARKETPLACE.md +8 -0
  22. package/README.md +53 -8
  23. package/bin/cli.js +27 -3
  24. package/config/domain_packs/biopharma.json +23 -0
  25. package/config/domain_packs/legal.json +19 -0
  26. package/config/domain_packs/quant.json +19 -0
  27. package/config/mcp-research-servers.json +7 -0
  28. package/config/mcp_launcher.py +48 -137
  29. package/config/opencode-snippet.json +58 -3
  30. package/config/searxng_mcp.py +42 -83
  31. package/extensions/pi/index.js +196 -28
  32. package/install.sh +20 -4
  33. package/package.json +40 -5
  34. package/plugins/antigravity/README.md +28 -0
  35. package/plugins/antigravity/agents/alpha-thesis.md +6 -0
  36. package/plugins/antigravity/agents/beta-antithesis.md +7 -0
  37. package/plugins/antigravity/agents/brainstormer.md +7 -0
  38. package/plugins/antigravity/agents/epistemic-auditor.md +5 -0
  39. package/plugins/antigravity/hooks.json +23 -0
  40. package/plugins/antigravity/mcp_config.json +33 -0
  41. package/plugins/antigravity/plugin.json +21 -0
  42. package/plugins/antigravity/rules/epistemic-integrity.md +6 -0
  43. package/plugins/antigravity/skills/brainstorming/SKILL.md +13 -0
  44. package/plugins/antigravity/skills/code_audit/SKILL.md +13 -0
  45. package/plugins/antigravity/skills/epistemic_search/SKILL.md +13 -0
  46. package/plugins/antigravity/skills/grilling/SKILL.md +13 -0
  47. package/plugins/antigravity/skills/oss_scout/SKILL.md +13 -0
  48. package/plugins/antigravity/skills/research_cache/SKILL.md +13 -0
  49. package/plugins/antigravity/skills/swarm_config/SKILL.md +13 -0
  50. package/plugins/codex/AGENTS.md.snippet +10 -0
  51. package/plugins/codex/README.md +37 -0
  52. package/plugins/codex/config.toml.snippet +28 -0
  53. package/plugins/codex/openai.yaml +24 -0
  54. package/plugins/codex/skills/brainstorming/SKILL.md +13 -0
  55. package/plugins/codex/skills/code_audit/SKILL.md +13 -0
  56. package/plugins/codex/skills/epistemic_search/SKILL.md +13 -0
  57. package/plugins/codex/skills/grilling/SKILL.md +13 -0
  58. package/plugins/codex/skills/oss_scout/SKILL.md +13 -0
  59. package/plugins/codex/skills/research_cache/SKILL.md +13 -0
  60. package/plugins/codex/skills/swarm_config/SKILL.md +13 -0
  61. package/plugins/gemini/GEMINI.md +15 -0
  62. package/plugins/gemini/README.md +19 -0
  63. package/plugins/gemini/commands/audit.toml +6 -0
  64. package/plugins/gemini/commands/brainstorming.toml +10 -0
  65. package/plugins/gemini/commands/grill.toml +6 -0
  66. package/plugins/gemini/commands/scout.toml +7 -0
  67. package/plugins/gemini/commands/swarm-config.toml +7 -0
  68. package/plugins/gemini/commands/swarm.toml +8 -0
  69. package/plugins/gemini/gemini-extension.json +38 -0
  70. package/plugins/gemini/hooks/hooks.json +11 -0
  71. package/plugins/gemini/skills/brainstorming/SKILL.md +13 -0
  72. package/plugins/gemini/skills/code_audit/SKILL.md +13 -0
  73. package/plugins/gemini/skills/epistemic_search/SKILL.md +13 -0
  74. package/plugins/gemini/skills/grilling/SKILL.md +13 -0
  75. package/plugins/gemini/skills/oss_scout/SKILL.md +13 -0
  76. package/plugins/gemini/skills/research_cache/SKILL.md +13 -0
  77. package/plugins/gemini/skills/swarm_config/SKILL.md +13 -0
  78. package/plugins/opencode/index.js +335 -118
  79. package/prompts/agent_brainstormer.md +97 -0
  80. package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
  81. package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
  82. package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
  83. package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
  84. package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
  85. package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
  86. package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
  87. package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
  88. package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
  89. package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
  90. package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
  91. package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
  92. package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
  93. package/runner/auctioneer.py +169 -0
  94. package/runner/auditor_engine.py +127 -12
  95. package/runner/claim_store.py +361 -0
  96. package/runner/claim_witness.py +183 -0
  97. package/runner/living_dossiers.py +355 -0
  98. package/runner/mcp_protocol.py +188 -0
  99. package/runner/mcp_server.py +567 -0
  100. package/runner/pcrb.py +212 -0
  101. package/runner/pcrb_verify.py +237 -0
  102. package/runner/refinement.py +335 -0
  103. package/runner/research_swarm.py +378 -94
  104. package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
  105. package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
  106. package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
  107. package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
  108. package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
  109. package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
  110. package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
  111. package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
  112. package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
  113. package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
  114. package/runner/tests/fixtures/auction_objective.json +35 -0
  115. package/runner/tests/fixtures/borderline_claims.json +27 -0
  116. package/runner/tests/fixtures/divergence_objectives.json +31 -0
  117. package/runner/tests/test_auction_order.py +173 -0
  118. package/runner/tests/test_claim_store.py +162 -0
  119. package/runner/tests/test_claim_witness.py +186 -0
  120. package/runner/tests/test_domain_packs.py +217 -0
  121. package/runner/tests/test_fleet_seam.py +113 -0
  122. package/runner/tests/test_living_dossiers.py +204 -0
  123. package/runner/tests/test_mcp_server.py +212 -0
  124. package/runner/tests/test_pcrb.py +240 -0
  125. package/runner/tests/test_refinement.py +255 -0
  126. package/runner/tests/test_swarm.py +173 -15
  127. package/scripts/auction_experiment.py +180 -0
  128. package/scripts/build_adapters.py +183 -0
  129. package/scripts/divergence_experiment.py +184 -0
  130. package/skills/brainstorming/SKILL.md +106 -0
  131. package/skills/brainstorming/__init__.py +1 -0
  132. package/skills/brainstorming/scripts/brainstorm.py +200 -0
  133. package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
  134. package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
  135. package/skills/swarm_config/SKILL.md +1 -1
  136. package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
  137. package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
  138. package/skills/swarm_config/configure.py +45 -11
@@ -0,0 +1,255 @@
1
+ #!/usr/bin/env python3
2
+ """Tests for the refinement-type checker and pure E(D) scoring.
3
+
4
+ Property-style checks are seeded hand-rolled loops (200 random claim sets)
5
+ rather than a hypothesis dependency — zero-dep constraint holds.
6
+ """
7
+
8
+ import random
9
+ import sys
10
+ import unittest
11
+ from pathlib import Path
12
+
13
+ PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
14
+ if str(PROJECT_ROOT) not in sys.path:
15
+ sys.path.insert(0, str(PROJECT_ROOT))
16
+
17
+ from runner.claim_witness import ( # noqa: E402
18
+ KIND_CLAIM,
19
+ KIND_HYPOTHESIS,
20
+ KIND_INFERENCE,
21
+ KIND_NEGATIVE_KNOWLEDGE,
22
+ STATUS_REJECTED,
23
+ TAG_HYPOTHESIS,
24
+ TAG_INFERRED,
25
+ TAG_NEGATIVE_KNOWLEDGE,
26
+ TAG_VERIFIED,
27
+ ClaimWitness,
28
+ )
29
+ from runner.refinement import ( # noqa: E402
30
+ LEGACY_CONSTITUTION,
31
+ Constitution,
32
+ check_invariants,
33
+ compute_epistemic_score,
34
+ compute_epistemic_score_from_claims,
35
+ )
36
+
37
+
38
+ def make_claim(**kw) -> ClaimWitness:
39
+ base = dict(
40
+ claim_id="C",
41
+ kind=KIND_CLAIM,
42
+ tag=TAG_VERIFIED,
43
+ statement="s",
44
+ source_hash="h",
45
+ verbatim_quote="q",
46
+ )
47
+ base.update(kw)
48
+ return ClaimWitness(**base)
49
+
50
+
51
+ class TestCheckInvariants(unittest.TestCase):
52
+ def test_verified_requires_hash_and_quote(self):
53
+ bad = make_claim(claim_id="C1", source_hash=None, verbatim_quote=None)
54
+ viols = check_invariants([bad])
55
+ rules = {v.rule for v in viols}
56
+ self.assertIn("VERIFIED_REQUIRES_HASH", rules)
57
+ self.assertIn("VERIFIED_REQUIRES_QUOTE", rules)
58
+
59
+ def test_inferred_requires_parents_and_logic(self):
60
+ bad = make_claim(
61
+ claim_id="I1",
62
+ kind=KIND_INFERENCE,
63
+ tag=TAG_INFERRED,
64
+ source_hash=None,
65
+ verbatim_quote=None,
66
+ parent_claims=[],
67
+ deductive_logic=None,
68
+ )
69
+ viols = check_invariants([bad])
70
+ rules = {v.rule for v in viols}
71
+ self.assertIn("INFERRED_REQUIRES_PARENTS", rules)
72
+ self.assertIn("INFERRED_REQUIRES_LOGIC", rules)
73
+
74
+ def test_hypothesis_requires_falsification(self):
75
+ bad = make_claim(
76
+ claim_id="H1",
77
+ kind=KIND_HYPOTHESIS,
78
+ tag=TAG_HYPOTHESIS,
79
+ source_hash=None,
80
+ verbatim_quote=None,
81
+ falsification=None,
82
+ )
83
+ viols = check_invariants([bad])
84
+ self.assertIn("HYPOTHESIS_REQUIRES_FALSIFICATION", {v.rule for v in viols})
85
+
86
+ def test_neg_knowledge_requires_query_and_finding(self):
87
+ bad = make_claim(
88
+ claim_id="N1",
89
+ kind=KIND_NEGATIVE_KNOWLEDGE,
90
+ tag=TAG_NEGATIVE_KNOWLEDGE,
91
+ source_hash=None,
92
+ verbatim_quote=None,
93
+ query=None,
94
+ finding=None,
95
+ )
96
+ viols = check_invariants([bad])
97
+ rules = {v.rule for v in viols}
98
+ self.assertIn("NEG_KNOWLEDGE_REQUIRES_QUERY", rules)
99
+ self.assertIn("NEG_KNOWLEDGE_REQUIRES_FINDING", rules)
100
+
101
+ def test_parent_cross_ref_resolution(self):
102
+ good_parent = make_claim(claim_id="C1")
103
+ orphan = make_claim(
104
+ claim_id="I1",
105
+ kind=KIND_INFERENCE,
106
+ tag=TAG_INFERRED,
107
+ source_hash=None,
108
+ verbatim_quote=None,
109
+ parent_claims=["DOES_NOT_EXIST"],
110
+ deductive_logic="because",
111
+ )
112
+ viols = check_invariants([good_parent, orphan])
113
+ self.assertIn("PARENT_UNRESOLVED", {v.rule for v in viols})
114
+
115
+ def test_clean_set_has_no_violations(self):
116
+ c1 = make_claim(claim_id="C1")
117
+ c2 = make_claim(
118
+ claim_id="I1",
119
+ kind=KIND_INFERENCE,
120
+ tag=TAG_INFERRED,
121
+ source_hash=None,
122
+ verbatim_quote=None,
123
+ parent_claims=["C1"],
124
+ deductive_logic="therefore",
125
+ )
126
+ self.assertEqual(check_invariants([c1, c2]), [])
127
+
128
+ def test_witness_check_with_hasher(self):
129
+ import tempfile
130
+ from skills.research_cache.hasher import SourceHasher
131
+
132
+ with tempfile.TemporaryDirectory() as tmp:
133
+ hasher = SourceHasher(Path(tmp))
134
+ content = "Verbatim sentence here."
135
+ digest = hasher.store_source(url="https://x", content=content)
136
+ good = make_claim(
137
+ claim_id="C1", source_hash=digest, verbatim_quote="Verbatim sentence"
138
+ )
139
+ bad = make_claim(
140
+ claim_id="C2", source_hash=digest, verbatim_quote="never in source"
141
+ )
142
+ viols = check_invariants([good, bad], hasher=hasher)
143
+ rules_by_id = {v.claim_id: v.rule for v in viols}
144
+ self.assertNotIn("C1", rules_by_id)
145
+ self.assertEqual(rules_by_id.get("C2"), "WITNESS_CHECK_FAILED")
146
+
147
+ def test_banned_domain_rule(self):
148
+ const = Constitution(banned_domains=["banned.example"])
149
+ bad = make_claim(claim_id="C1", source_url="https://banned.example/paper")
150
+ viols = check_invariants([bad], constitution=const)
151
+ self.assertIn("BANNED_DOMAIN", {v.rule for v in viols})
152
+
153
+
154
+ class TestComputeEpistemicScore(unittest.TestCase):
155
+ def test_legacy_formula_exact(self):
156
+ # 1 verified, 1 inferred, 1 hypothesis, 1 rejected, 1 neg_knowledge
157
+ # raw = (1.0*1 + 0.5*1 - 2.5*1) / (1+1+1+1) = (1+0.5-2.5)/4 = -1/4 = -0.25
158
+ # clamped to 0.0
159
+ score, br = compute_epistemic_score(
160
+ {"verified": 1, "rejected": 1, "inferred": 1, "hypotheses": 1, "neg_knowledge": 1}
161
+ )
162
+ self.assertEqual(score, 0.0)
163
+ self.assertEqual(br["verified_passed"], 1)
164
+ self.assertEqual(br["unverified_rejected"], 1)
165
+ self.assertEqual(br["negative_knowledge_count"], 1)
166
+
167
+ def test_legacy_formula_all_verified(self):
168
+ # 2 verified, 0 else: raw = 2/2 = 1.0
169
+ score, _ = compute_epistemic_score(
170
+ {"verified": 2, "rejected": 0, "inferred": 0, "hypotheses": 0, "neg_knowledge": 0}
171
+ )
172
+ self.assertEqual(score, 1.0)
173
+
174
+ def test_score_bounded(self):
175
+ for counts in (
176
+ {"verified": 0, "rejected": 0, "inferred": 0, "hypotheses": 0, "neg_knowledge": 0},
177
+ {"verified": 0, "rejected": 100, "inferred": 0, "hypotheses": 0, "neg_knowledge": 0},
178
+ {"verified": 100, "rejected": 0, "inferred": 0, "hypotheses": 0, "neg_knowledge": 100},
179
+ ):
180
+ score, _ = compute_epistemic_score(counts)
181
+ self.assertGreaterEqual(score, 0.0)
182
+ self.assertLessEqual(score, 1.0)
183
+
184
+ def test_deterministic(self):
185
+ counts = {"verified": 3, "rejected": 1, "inferred": 2, "hypotheses": 1, "neg_knowledge": 2}
186
+ s1, b1 = compute_epistemic_score(counts)
187
+ s2, b2 = compute_epistemic_score(counts)
188
+ self.assertEqual(s1, s2)
189
+ self.assertEqual(b1, b2)
190
+
191
+ def test_monotone_rejection_penalty(self):
192
+ base = {"verified": 5, "rejected": 0, "inferred": 0, "hypotheses": 0, "neg_knowledge": 0}
193
+ more_rejects = dict(base, rejected=3)
194
+ s0, _ = compute_epistemic_score(base)
195
+ s1, _ = compute_epistemic_score(more_rejects)
196
+ self.assertLessEqual(s1, s0)
197
+
198
+ def test_zero_assertions_clamps(self):
199
+ score, br = compute_epistemic_score(
200
+ {"verified": 0, "rejected": 0, "inferred": 0, "hypotheses": 0, "neg_knowledge": 0}
201
+ )
202
+ self.assertEqual(score, 0.0)
203
+ self.assertEqual(br["total_assertions"], 1)
204
+
205
+ def test_property_loops_200_random_claim_sets(self):
206
+ rng = random.Random(0xF00D)
207
+ for _ in range(200):
208
+ counts = {
209
+ "verified": rng.randint(0, 20),
210
+ "rejected": rng.randint(0, 10),
211
+ "inferred": rng.randint(0, 10),
212
+ "hypotheses": rng.randint(0, 10),
213
+ "neg_knowledge": rng.randint(0, 5),
214
+ }
215
+ score, _br = compute_epistemic_score(counts)
216
+ self.assertGreaterEqual(score, 0.0, counts)
217
+ self.assertLessEqual(score, 1.0, counts)
218
+ # Determinism
219
+ score2, _ = compute_epistemic_score(counts)
220
+ self.assertEqual(score, score2)
221
+ # More rejections never improve the score
222
+ worse = dict(counts, rejected=counts["rejected"] + 1)
223
+ worse_score, _ = compute_epistemic_score(worse)
224
+ self.assertLessEqual(worse_score, score, (counts, worse))
225
+
226
+
227
+ class TestConstitution(unittest.TestCase):
228
+ def test_legacy_default_is_flat(self):
229
+ self.assertEqual(LEGACY_CONSTITUTION.weight_for(None), 1.0)
230
+ self.assertEqual(LEGACY_CONSTITUTION.weight_for("PREPRINT"), 1.0)
231
+
232
+ def test_tier_weights_override(self):
233
+ const = Constitution(tier_weights={"PREPRINT": 0.3, "__default__": 1.0})
234
+ self.assertEqual(const.weight_for("PREPRINT"), 0.3)
235
+ self.assertEqual(const.weight_for("PEER_REVIEWED"), 1.0)
236
+
237
+ def test_tier_weighted_claims_path(self):
238
+ # Legacy: 2 verified = 1.0
239
+ legacy_claims = [make_claim(claim_id=f"C{i}") for i in range(2)]
240
+ s_legacy, _ = compute_epistemic_score_from_claims(legacy_claims)
241
+ self.assertEqual(s_legacy, 1.0)
242
+
243
+ # Biopharma-ish: preprints downgraded to 0.3
244
+ const = Constitution(tier_weights={"PREPRINT": 0.3, "__default__": 1.0})
245
+ preprints = [
246
+ make_claim(claim_id=f"C{i}", tier="PREPRINT") for i in range(2)
247
+ ]
248
+ s_tiered, br = compute_epistemic_score_from_claims(preprints, constitution=const)
249
+ # weight = 0.3*2 = 0.6, assertions = 2, raw = 0.6/2 = 0.3
250
+ self.assertEqual(s_tiered, 0.3)
251
+ self.assertAlmostEqual(br["verified_weight"], 0.6, places=3)
252
+
253
+
254
+ if __name__ == "__main__":
255
+ unittest.main()
@@ -20,6 +20,7 @@ from runner.state_machine import ResearchStateMachine, SessionStatus, ScopeStatu
20
20
  from runner.auditor_engine import EpistemicAuditorEngine
21
21
  from runner.research_swarm import SwarmRunner
22
22
 
23
+
23
24
  class TestEpistemicSwarm(unittest.TestCase):
24
25
  def setUp(self):
25
26
  self.test_dir = Path(tempfile.mkdtemp(prefix="epistemic_test_"))
@@ -38,7 +39,7 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
38
39
  shash = self.hasher.store_source(
39
40
  url="https://arxiv.org/abs/2203.15556",
40
41
  content=content,
41
- title="Chinchilla Scaling Laws"
42
+ title="Chinchilla Scaling Laws",
42
43
  )
43
44
  self.assertTrue(len(shash) == 64)
44
45
 
@@ -57,7 +58,8 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
57
58
 
58
59
  # 3. Fabricated quote rejection
59
60
  verified_fake, conf_fake, _ = self.hasher.verify_quote(
60
- shash, "the 70B parameter model was trained on 500 quadrillion tokens by aliens."
61
+ shash,
62
+ "the 70B parameter model was trained on 500 quadrillion tokens by aliens.",
61
63
  )
62
64
  self.assertFalse(verified_fake)
63
65
  self.assertLess(conf_fake, 0.8)
@@ -68,13 +70,13 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
68
70
  {
69
71
  "scope_id": "scope_01_prover",
70
72
  "title": "Prover Benchmarks",
71
- "dependencies": []
73
+ "dependencies": [],
72
74
  },
73
75
  {
74
76
  "scope_id": "scope_02_recursion",
75
77
  "title": "Recursive Verification",
76
- "dependencies": ["scope_01_prover"]
77
- }
78
+ "dependencies": ["scope_01_prover"],
79
+ },
78
80
  ]
79
81
  self.state_machine.set_scopes(scopes)
80
82
 
@@ -96,7 +98,7 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
96
98
  shash = self.hasher.store_source(
97
99
  url="https://benchmark.org/zk",
98
100
  content="FPGA prover executes Poseidon in 184ms.",
99
- title="ZK Benchmarks"
101
+ title="ZK Benchmarks",
100
102
  )
101
103
 
102
104
  # 2. Initialize scope
@@ -113,17 +115,17 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
113
115
  "tag": "VERIFIED",
114
116
  "statement": "Poseidon prover executes in 184ms",
115
117
  "source_hash": shash,
116
- "verbatim_quote": "FPGA prover executes Poseidon in 184ms."
118
+ "verbatim_quote": "FPGA prover executes Poseidon in 184ms.",
117
119
  },
118
120
  {
119
121
  "claim_id": "A2",
120
122
  "tag": "VERIFIED",
121
123
  "statement": "Hallucinated claim that does not exist in source",
122
124
  "source_hash": shash,
123
- "verbatim_quote": "This string does not exist anywhere in the text."
124
- }
125
+ "verbatim_quote": "This string does not exist anywhere in the text.",
126
+ },
125
127
  ],
126
- "negative_knowledge": [{"query": "q1", "finding": "None"}]
128
+ "negative_knowledge": [{"query": "q1", "finding": "None"}],
127
129
  }
128
130
 
129
131
  # 4. Create Beta Dossier
@@ -134,10 +136,10 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
134
136
  "methodological_critiques": [
135
137
  {
136
138
  "target_assertion": "Poseidon prover executes in 184ms",
137
- "critique": "Benchmark excludes PCIe host bus latency"
139
+ "critique": "Benchmark excludes PCIe host bus latency",
138
140
  }
139
141
  ],
140
- "negative_knowledge": []
142
+ "negative_knowledge": [],
141
143
  }
142
144
 
143
145
  self.state_machine.record_agent_completion("scope_test", "alpha", alpha_dossier)
@@ -189,7 +191,9 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
189
191
 
190
192
  for ext_path in pkg["pi"].get("extensions", []):
191
193
  resolved = (PROJECT_ROOT / ext_path).resolve()
192
- self.assertTrue(resolved.exists(), f"Pi extension path not found: {ext_path}")
194
+ self.assertTrue(
195
+ resolved.exists(), f"Pi extension path not found: {ext_path}"
196
+ )
193
197
 
194
198
  # 2. OpenCode plugin validation
195
199
  self.assertIn("opencode", pkg.get("keywords", []))
@@ -203,6 +207,7 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
203
207
 
204
208
  def test_config_manager_load_and_save(self):
205
209
  from skills.swarm_config.configure import load_config, save_config
210
+
206
211
  cfg = load_config(str(self.test_dir))
207
212
  self.assertEqual(cfg["search_engine"], "duckduckgo")
208
213
  self.assertEqual(cfg["mode"], "research")
@@ -237,8 +242,161 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
237
242
  content = f.read()
238
243
  self.assertIn("Open-Source Software Discovery", content)
239
244
 
245
+ def test_mock_brainstorm_mode(self):
246
+ runner = SwarmRunner(base_dir=self.test_dir, mock_mode=True, mode="brainstorm")
247
+ runner.run_swarm("Where do we go from here?")
240
248
 
241
- if __name__ == "__main__":
242
- unittest.main()
249
+ brainstorm_report = self.test_dir / "brainstorm_report.md"
250
+ self.assertTrue(brainstorm_report.exists())
251
+ with open(brainstorm_report, "r") as f:
252
+ content = f.read()
253
+ self.assertIn("Lateral Brainstorm", content)
254
+
255
+ def test_brainstorm_prompt_and_skill_exist(self):
256
+ prompt = PROJECT_ROOT / "prompts" / "agent_brainstormer.md"
257
+ skill = PROJECT_ROOT / "skills" / "brainstorming" / "SKILL.md"
258
+ scaffold = (
259
+ PROJECT_ROOT / "skills" / "brainstorming" / "scripts" / "brainstorm.py"
260
+ )
261
+ self.assertTrue(prompt.exists())
262
+ self.assertTrue(skill.exists())
263
+ self.assertTrue(scaffold.exists())
264
+ text = skill.read_text(encoding="utf-8")
265
+ self.assertIn("name: brainstorming", text)
266
+ self.assertIn("HYPOTHESIS", text)
267
+
268
+ def test_universal_adapter_bundles_exist(self):
269
+ # AntiGravity
270
+ self.assertTrue((PROJECT_ROOT / "plugins/antigravity/plugin.json").exists())
271
+ self.assertTrue((PROJECT_ROOT / "plugins/antigravity/mcp_config.json").exists())
272
+ self.assertTrue((PROJECT_ROOT / "plugins/antigravity/hooks.json").exists())
273
+ self.assertTrue(
274
+ (
275
+ PROJECT_ROOT / "plugins/antigravity/skills/brainstorming/SKILL.md"
276
+ ).exists()
277
+ )
278
+ self.assertTrue(
279
+ (PROJECT_ROOT / "plugins/antigravity/agents/brainstormer.md").exists()
280
+ )
281
+ # Gemini
282
+ self.assertTrue(
283
+ (PROJECT_ROOT / "plugins/gemini/gemini-extension.json").exists()
284
+ )
285
+ self.assertTrue((PROJECT_ROOT / "plugins/gemini/GEMINI.md").exists())
286
+ self.assertTrue(
287
+ (PROJECT_ROOT / "plugins/gemini/commands/brainstorming.toml").exists()
288
+ )
289
+ self.assertTrue(
290
+ (PROJECT_ROOT / "plugins/gemini/skills/brainstorming/SKILL.md").exists()
291
+ )
292
+ # Codex
293
+ self.assertTrue((PROJECT_ROOT / "plugins/codex/openai.yaml").exists())
294
+ self.assertTrue((PROJECT_ROOT / "plugins/codex/config.toml.snippet").exists())
295
+ self.assertTrue(
296
+ (PROJECT_ROOT / "plugins/codex/skills/brainstorming/SKILL.md").exists()
297
+ )
298
+ self.assertTrue(
299
+ (PROJECT_ROOT / ".agents/skills/brainstorming/SKILL.md").exists()
300
+ )
301
+ # OMP (oh-my-pi)
302
+ self.assertTrue((PROJECT_ROOT / ".omp/commands/brainstorming.md").exists())
303
+ self.assertTrue((PROJECT_ROOT / ".omp/prompts/brainstorming.md").exists())
304
+ self.assertTrue((PROJECT_ROOT / ".omp/SYSTEM.md").exists())
305
+ self.assertTrue(
306
+ (PROJECT_ROOT / ".omp/hooks/pre/epistemic-redirect.ts").exists()
307
+ )
308
+ # package.json omp block mirrors pi block
309
+ with open(PROJECT_ROOT / "package.json", "r", encoding="utf-8") as f:
310
+ pkg = json.load(f)
311
+ self.assertIn("omp", pkg)
312
+ self.assertIn("./extensions/pi/index.js", pkg["omp"].get("extensions", []))
313
+ self.assertIn("./skills/brainstorming", pkg["omp"].get("skills", []))
314
+ self.assertIn("./skills/brainstorming", pkg["pi"].get("skills", []))
315
+
316
+ def test_opencode_and_pi_extension_interfaces(self):
317
+ import subprocess
318
+
319
+ # 1. Test OpenCode Plugin registration (A2b: thin forwarders + both
320
+ # name exports — canonical IUMBTEMS_TOOL_NAMES and the deprecated
321
+ # IUMBEMS_TOOL_NAMES alias).
322
+ node_code_oc = """
323
+ import plugin, { IUMBTEMS_TOOL_NAMES, IUMBEMS_TOOL_NAMES } from "./plugins/opencode/index.js";
324
+ const tools = plugin.server();
325
+ const names = tools.map(t => t.name);
326
+ console.log(JSON.stringify({
327
+ names,
328
+ canonical: IUMBTEMS_TOOL_NAMES,
329
+ alias: IUMBEMS_TOOL_NAMES
330
+ }));
331
+ """
332
+ res_oc = subprocess.run(
333
+ ["node", "--input-type=module", "-e", node_code_oc],
334
+ capture_output=True,
335
+ text=True,
336
+ cwd=str(PROJECT_ROOT),
337
+ )
338
+ self.assertEqual(
339
+ res_oc.returncode, 0, f"OpenCode plugin test failed: {res_oc.stderr}"
340
+ )
341
+ lines = [
342
+ line.strip()
343
+ for line in res_oc.stdout.strip().split("\n")
344
+ if line.strip().startswith("{")
345
+ ]
346
+ data_oc = json.loads(lines[-1])
347
+ tools = data_oc["names"]
348
+ # Canonical + deprecated alias must both exist and agree.
349
+ self.assertEqual(data_oc["canonical"], data_oc["alias"])
350
+ self.assertEqual(sorted(data_oc["canonical"]), sorted(tools))
351
+ expected_tools = [
352
+ "iumbtems_config",
353
+ "iumbtems_swarm_research",
354
+ "iumbtems_code_audit",
355
+ "iumbtems_oss_scout",
356
+ "iumbtems_brainstorm",
357
+ "iumbtems_verify_quote",
358
+ "iumbtems_socratic_frontier",
359
+ "iumbtems_reindex_claims",
360
+ "iumbtems_report_retraction",
361
+ "iumbtems_check_staleness",
362
+ "iumbtems_set_domain_pack",
363
+ "iumbtems_export_brief",
364
+ "iumbtems_verify_brief",
365
+ ]
366
+ for t in expected_tools:
367
+ self.assertIn(t, tools)
368
+
369
+ # 2. Test Pi Extension registration
370
+ node_code_pi = """
371
+ import initPi from "./extensions/pi/index.js";
372
+ const commands = [];
373
+ const tools = [];
374
+ initPi({
375
+ registerCommand: (name) => commands.push(name),
376
+ registerTool: (def) => tools.push(def.name)
377
+ });
378
+ console.log(JSON.stringify({ commands, tools }));
379
+ """
380
+ res_pi = subprocess.run(
381
+ ["node", "--input-type=module", "-e", node_code_pi],
382
+ capture_output=True,
383
+ text=True,
384
+ cwd=str(PROJECT_ROOT),
385
+ )
386
+ self.assertEqual(
387
+ res_pi.returncode, 0, f"Pi extension test failed: {res_pi.stderr}"
388
+ )
389
+ lines = [
390
+ line.strip()
391
+ for line in res_pi.stdout.strip().split("\n")
392
+ if line.strip().startswith("{")
393
+ ]
394
+ data = json.loads(lines[-1])
395
+ for c in ["swarm", "grill", "swarm-config", "audit", "scout", "brainstorming"]:
396
+ self.assertIn(c, data["commands"])
397
+ for t in ["iumbtems_verify_quote", "iumbtems_config", "iumbtems_brainstorm"]:
398
+ self.assertIn(t, data["tools"])
243
399
 
244
400
 
401
+ if __name__ == "__main__":
402
+ unittest.main()