@heretek-ai/epistemic-swarm 0.2.2 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/skills/brainstorming/SKILL.md +13 -0
- package/.agents/skills/code_audit/SKILL.md +13 -0
- package/.agents/skills/epistemic_search/SKILL.md +13 -0
- package/.agents/skills/grilling/SKILL.md +13 -0
- package/.agents/skills/oss_scout/SKILL.md +13 -0
- package/.agents/skills/research_cache/SKILL.md +13 -0
- package/.agents/skills/swarm_config/SKILL.md +13 -0
- package/.claude-plugin/plugin.json +15 -5
- package/.omp/README.md +39 -0
- package/.omp/SYSTEM.md +12 -0
- package/.omp/commands/audit.md +10 -0
- package/.omp/commands/brainstorming.md +12 -0
- package/.omp/commands/grill.md +9 -0
- package/.omp/commands/scout.md +10 -0
- package/.omp/commands/swarm-config.md +10 -0
- package/.omp/commands/swarm.md +9 -0
- package/.omp/hooks/post/epistemic-audit.ts +16 -0
- package/.omp/hooks/pre/epistemic-redirect.ts +21 -0
- package/.omp/prompts/brainstorming.md +8 -0
- package/.omp/prompts/swarm.md +8 -0
- package/MARKETPLACE.md +8 -0
- package/README.md +53 -8
- package/bin/cli.js +27 -3
- package/config/domain_packs/biopharma.json +23 -0
- package/config/domain_packs/legal.json +19 -0
- package/config/domain_packs/quant.json +19 -0
- package/config/mcp-research-servers.json +7 -0
- package/config/mcp_launcher.py +48 -137
- package/config/opencode-snippet.json +58 -3
- package/config/searxng_mcp.py +42 -83
- package/extensions/pi/index.js +196 -28
- package/install.sh +20 -4
- package/package.json +40 -5
- package/plugins/antigravity/README.md +28 -0
- package/plugins/antigravity/agents/alpha-thesis.md +6 -0
- package/plugins/antigravity/agents/beta-antithesis.md +7 -0
- package/plugins/antigravity/agents/brainstormer.md +7 -0
- package/plugins/antigravity/agents/epistemic-auditor.md +5 -0
- package/plugins/antigravity/hooks.json +23 -0
- package/plugins/antigravity/mcp_config.json +33 -0
- package/plugins/antigravity/plugin.json +21 -0
- package/plugins/antigravity/rules/epistemic-integrity.md +6 -0
- package/plugins/antigravity/skills/brainstorming/SKILL.md +13 -0
- package/plugins/antigravity/skills/code_audit/SKILL.md +13 -0
- package/plugins/antigravity/skills/epistemic_search/SKILL.md +13 -0
- package/plugins/antigravity/skills/grilling/SKILL.md +13 -0
- package/plugins/antigravity/skills/oss_scout/SKILL.md +13 -0
- package/plugins/antigravity/skills/research_cache/SKILL.md +13 -0
- package/plugins/antigravity/skills/swarm_config/SKILL.md +13 -0
- package/plugins/codex/AGENTS.md.snippet +10 -0
- package/plugins/codex/README.md +37 -0
- package/plugins/codex/config.toml.snippet +28 -0
- package/plugins/codex/openai.yaml +24 -0
- package/plugins/codex/skills/brainstorming/SKILL.md +13 -0
- package/plugins/codex/skills/code_audit/SKILL.md +13 -0
- package/plugins/codex/skills/epistemic_search/SKILL.md +13 -0
- package/plugins/codex/skills/grilling/SKILL.md +13 -0
- package/plugins/codex/skills/oss_scout/SKILL.md +13 -0
- package/plugins/codex/skills/research_cache/SKILL.md +13 -0
- package/plugins/codex/skills/swarm_config/SKILL.md +13 -0
- package/plugins/gemini/GEMINI.md +15 -0
- package/plugins/gemini/README.md +19 -0
- package/plugins/gemini/commands/audit.toml +6 -0
- package/plugins/gemini/commands/brainstorming.toml +10 -0
- package/plugins/gemini/commands/grill.toml +6 -0
- package/plugins/gemini/commands/scout.toml +7 -0
- package/plugins/gemini/commands/swarm-config.toml +7 -0
- package/plugins/gemini/commands/swarm.toml +8 -0
- package/plugins/gemini/gemini-extension.json +38 -0
- package/plugins/gemini/hooks/hooks.json +11 -0
- package/plugins/gemini/skills/brainstorming/SKILL.md +13 -0
- package/plugins/gemini/skills/code_audit/SKILL.md +13 -0
- package/plugins/gemini/skills/epistemic_search/SKILL.md +13 -0
- package/plugins/gemini/skills/grilling/SKILL.md +13 -0
- package/plugins/gemini/skills/oss_scout/SKILL.md +13 -0
- package/plugins/gemini/skills/research_cache/SKILL.md +13 -0
- package/plugins/gemini/skills/swarm_config/SKILL.md +13 -0
- package/plugins/opencode/index.js +335 -118
- package/prompts/agent_brainstormer.md +97 -0
- package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
- package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
- package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
- package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
- package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
- package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
- package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
- package/runner/auctioneer.py +169 -0
- package/runner/auditor_engine.py +127 -12
- package/runner/claim_store.py +361 -0
- package/runner/claim_witness.py +183 -0
- package/runner/living_dossiers.py +355 -0
- package/runner/mcp_protocol.py +188 -0
- package/runner/mcp_server.py +567 -0
- package/runner/pcrb.py +212 -0
- package/runner/pcrb_verify.py +237 -0
- package/runner/refinement.py +335 -0
- package/runner/research_swarm.py +378 -94
- package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
- package/runner/tests/fixtures/auction_objective.json +35 -0
- package/runner/tests/fixtures/borderline_claims.json +27 -0
- package/runner/tests/fixtures/divergence_objectives.json +31 -0
- package/runner/tests/test_auction_order.py +173 -0
- package/runner/tests/test_claim_store.py +162 -0
- package/runner/tests/test_claim_witness.py +186 -0
- package/runner/tests/test_domain_packs.py +217 -0
- package/runner/tests/test_fleet_seam.py +113 -0
- package/runner/tests/test_living_dossiers.py +204 -0
- package/runner/tests/test_mcp_server.py +212 -0
- package/runner/tests/test_pcrb.py +240 -0
- package/runner/tests/test_refinement.py +255 -0
- package/runner/tests/test_swarm.py +173 -15
- package/scripts/auction_experiment.py +180 -0
- package/scripts/build_adapters.py +183 -0
- package/scripts/divergence_experiment.py +184 -0
- package/skills/brainstorming/SKILL.md +106 -0
- package/skills/brainstorming/__init__.py +1 -0
- package/skills/brainstorming/scripts/brainstorm.py +200 -0
- package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
- package/skills/swarm_config/SKILL.md +1 -1
- package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
- package/skills/swarm_config/configure.py +45 -11
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Tests for the refinement-type checker and pure E(D) scoring.
|
|
3
|
+
|
|
4
|
+
Property-style checks are seeded hand-rolled loops (200 random claim sets)
|
|
5
|
+
rather than a hypothesis dependency — zero-dep constraint holds.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import random
|
|
9
|
+
import sys
|
|
10
|
+
import unittest
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
|
|
14
|
+
if str(PROJECT_ROOT) not in sys.path:
|
|
15
|
+
sys.path.insert(0, str(PROJECT_ROOT))
|
|
16
|
+
|
|
17
|
+
from runner.claim_witness import ( # noqa: E402
|
|
18
|
+
KIND_CLAIM,
|
|
19
|
+
KIND_HYPOTHESIS,
|
|
20
|
+
KIND_INFERENCE,
|
|
21
|
+
KIND_NEGATIVE_KNOWLEDGE,
|
|
22
|
+
STATUS_REJECTED,
|
|
23
|
+
TAG_HYPOTHESIS,
|
|
24
|
+
TAG_INFERRED,
|
|
25
|
+
TAG_NEGATIVE_KNOWLEDGE,
|
|
26
|
+
TAG_VERIFIED,
|
|
27
|
+
ClaimWitness,
|
|
28
|
+
)
|
|
29
|
+
from runner.refinement import ( # noqa: E402
|
|
30
|
+
LEGACY_CONSTITUTION,
|
|
31
|
+
Constitution,
|
|
32
|
+
check_invariants,
|
|
33
|
+
compute_epistemic_score,
|
|
34
|
+
compute_epistemic_score_from_claims,
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def make_claim(**kw) -> ClaimWitness:
|
|
39
|
+
base = dict(
|
|
40
|
+
claim_id="C",
|
|
41
|
+
kind=KIND_CLAIM,
|
|
42
|
+
tag=TAG_VERIFIED,
|
|
43
|
+
statement="s",
|
|
44
|
+
source_hash="h",
|
|
45
|
+
verbatim_quote="q",
|
|
46
|
+
)
|
|
47
|
+
base.update(kw)
|
|
48
|
+
return ClaimWitness(**base)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class TestCheckInvariants(unittest.TestCase):
|
|
52
|
+
def test_verified_requires_hash_and_quote(self):
|
|
53
|
+
bad = make_claim(claim_id="C1", source_hash=None, verbatim_quote=None)
|
|
54
|
+
viols = check_invariants([bad])
|
|
55
|
+
rules = {v.rule for v in viols}
|
|
56
|
+
self.assertIn("VERIFIED_REQUIRES_HASH", rules)
|
|
57
|
+
self.assertIn("VERIFIED_REQUIRES_QUOTE", rules)
|
|
58
|
+
|
|
59
|
+
def test_inferred_requires_parents_and_logic(self):
|
|
60
|
+
bad = make_claim(
|
|
61
|
+
claim_id="I1",
|
|
62
|
+
kind=KIND_INFERENCE,
|
|
63
|
+
tag=TAG_INFERRED,
|
|
64
|
+
source_hash=None,
|
|
65
|
+
verbatim_quote=None,
|
|
66
|
+
parent_claims=[],
|
|
67
|
+
deductive_logic=None,
|
|
68
|
+
)
|
|
69
|
+
viols = check_invariants([bad])
|
|
70
|
+
rules = {v.rule for v in viols}
|
|
71
|
+
self.assertIn("INFERRED_REQUIRES_PARENTS", rules)
|
|
72
|
+
self.assertIn("INFERRED_REQUIRES_LOGIC", rules)
|
|
73
|
+
|
|
74
|
+
def test_hypothesis_requires_falsification(self):
|
|
75
|
+
bad = make_claim(
|
|
76
|
+
claim_id="H1",
|
|
77
|
+
kind=KIND_HYPOTHESIS,
|
|
78
|
+
tag=TAG_HYPOTHESIS,
|
|
79
|
+
source_hash=None,
|
|
80
|
+
verbatim_quote=None,
|
|
81
|
+
falsification=None,
|
|
82
|
+
)
|
|
83
|
+
viols = check_invariants([bad])
|
|
84
|
+
self.assertIn("HYPOTHESIS_REQUIRES_FALSIFICATION", {v.rule for v in viols})
|
|
85
|
+
|
|
86
|
+
def test_neg_knowledge_requires_query_and_finding(self):
|
|
87
|
+
bad = make_claim(
|
|
88
|
+
claim_id="N1",
|
|
89
|
+
kind=KIND_NEGATIVE_KNOWLEDGE,
|
|
90
|
+
tag=TAG_NEGATIVE_KNOWLEDGE,
|
|
91
|
+
source_hash=None,
|
|
92
|
+
verbatim_quote=None,
|
|
93
|
+
query=None,
|
|
94
|
+
finding=None,
|
|
95
|
+
)
|
|
96
|
+
viols = check_invariants([bad])
|
|
97
|
+
rules = {v.rule for v in viols}
|
|
98
|
+
self.assertIn("NEG_KNOWLEDGE_REQUIRES_QUERY", rules)
|
|
99
|
+
self.assertIn("NEG_KNOWLEDGE_REQUIRES_FINDING", rules)
|
|
100
|
+
|
|
101
|
+
def test_parent_cross_ref_resolution(self):
|
|
102
|
+
good_parent = make_claim(claim_id="C1")
|
|
103
|
+
orphan = make_claim(
|
|
104
|
+
claim_id="I1",
|
|
105
|
+
kind=KIND_INFERENCE,
|
|
106
|
+
tag=TAG_INFERRED,
|
|
107
|
+
source_hash=None,
|
|
108
|
+
verbatim_quote=None,
|
|
109
|
+
parent_claims=["DOES_NOT_EXIST"],
|
|
110
|
+
deductive_logic="because",
|
|
111
|
+
)
|
|
112
|
+
viols = check_invariants([good_parent, orphan])
|
|
113
|
+
self.assertIn("PARENT_UNRESOLVED", {v.rule for v in viols})
|
|
114
|
+
|
|
115
|
+
def test_clean_set_has_no_violations(self):
|
|
116
|
+
c1 = make_claim(claim_id="C1")
|
|
117
|
+
c2 = make_claim(
|
|
118
|
+
claim_id="I1",
|
|
119
|
+
kind=KIND_INFERENCE,
|
|
120
|
+
tag=TAG_INFERRED,
|
|
121
|
+
source_hash=None,
|
|
122
|
+
verbatim_quote=None,
|
|
123
|
+
parent_claims=["C1"],
|
|
124
|
+
deductive_logic="therefore",
|
|
125
|
+
)
|
|
126
|
+
self.assertEqual(check_invariants([c1, c2]), [])
|
|
127
|
+
|
|
128
|
+
def test_witness_check_with_hasher(self):
|
|
129
|
+
import tempfile
|
|
130
|
+
from skills.research_cache.hasher import SourceHasher
|
|
131
|
+
|
|
132
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
133
|
+
hasher = SourceHasher(Path(tmp))
|
|
134
|
+
content = "Verbatim sentence here."
|
|
135
|
+
digest = hasher.store_source(url="https://x", content=content)
|
|
136
|
+
good = make_claim(
|
|
137
|
+
claim_id="C1", source_hash=digest, verbatim_quote="Verbatim sentence"
|
|
138
|
+
)
|
|
139
|
+
bad = make_claim(
|
|
140
|
+
claim_id="C2", source_hash=digest, verbatim_quote="never in source"
|
|
141
|
+
)
|
|
142
|
+
viols = check_invariants([good, bad], hasher=hasher)
|
|
143
|
+
rules_by_id = {v.claim_id: v.rule for v in viols}
|
|
144
|
+
self.assertNotIn("C1", rules_by_id)
|
|
145
|
+
self.assertEqual(rules_by_id.get("C2"), "WITNESS_CHECK_FAILED")
|
|
146
|
+
|
|
147
|
+
def test_banned_domain_rule(self):
|
|
148
|
+
const = Constitution(banned_domains=["banned.example"])
|
|
149
|
+
bad = make_claim(claim_id="C1", source_url="https://banned.example/paper")
|
|
150
|
+
viols = check_invariants([bad], constitution=const)
|
|
151
|
+
self.assertIn("BANNED_DOMAIN", {v.rule for v in viols})
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
class TestComputeEpistemicScore(unittest.TestCase):
|
|
155
|
+
def test_legacy_formula_exact(self):
|
|
156
|
+
# 1 verified, 1 inferred, 1 hypothesis, 1 rejected, 1 neg_knowledge
|
|
157
|
+
# raw = (1.0*1 + 0.5*1 - 2.5*1) / (1+1+1+1) = (1+0.5-2.5)/4 = -1/4 = -0.25
|
|
158
|
+
# clamped to 0.0
|
|
159
|
+
score, br = compute_epistemic_score(
|
|
160
|
+
{"verified": 1, "rejected": 1, "inferred": 1, "hypotheses": 1, "neg_knowledge": 1}
|
|
161
|
+
)
|
|
162
|
+
self.assertEqual(score, 0.0)
|
|
163
|
+
self.assertEqual(br["verified_passed"], 1)
|
|
164
|
+
self.assertEqual(br["unverified_rejected"], 1)
|
|
165
|
+
self.assertEqual(br["negative_knowledge_count"], 1)
|
|
166
|
+
|
|
167
|
+
def test_legacy_formula_all_verified(self):
|
|
168
|
+
# 2 verified, 0 else: raw = 2/2 = 1.0
|
|
169
|
+
score, _ = compute_epistemic_score(
|
|
170
|
+
{"verified": 2, "rejected": 0, "inferred": 0, "hypotheses": 0, "neg_knowledge": 0}
|
|
171
|
+
)
|
|
172
|
+
self.assertEqual(score, 1.0)
|
|
173
|
+
|
|
174
|
+
def test_score_bounded(self):
|
|
175
|
+
for counts in (
|
|
176
|
+
{"verified": 0, "rejected": 0, "inferred": 0, "hypotheses": 0, "neg_knowledge": 0},
|
|
177
|
+
{"verified": 0, "rejected": 100, "inferred": 0, "hypotheses": 0, "neg_knowledge": 0},
|
|
178
|
+
{"verified": 100, "rejected": 0, "inferred": 0, "hypotheses": 0, "neg_knowledge": 100},
|
|
179
|
+
):
|
|
180
|
+
score, _ = compute_epistemic_score(counts)
|
|
181
|
+
self.assertGreaterEqual(score, 0.0)
|
|
182
|
+
self.assertLessEqual(score, 1.0)
|
|
183
|
+
|
|
184
|
+
def test_deterministic(self):
|
|
185
|
+
counts = {"verified": 3, "rejected": 1, "inferred": 2, "hypotheses": 1, "neg_knowledge": 2}
|
|
186
|
+
s1, b1 = compute_epistemic_score(counts)
|
|
187
|
+
s2, b2 = compute_epistemic_score(counts)
|
|
188
|
+
self.assertEqual(s1, s2)
|
|
189
|
+
self.assertEqual(b1, b2)
|
|
190
|
+
|
|
191
|
+
def test_monotone_rejection_penalty(self):
|
|
192
|
+
base = {"verified": 5, "rejected": 0, "inferred": 0, "hypotheses": 0, "neg_knowledge": 0}
|
|
193
|
+
more_rejects = dict(base, rejected=3)
|
|
194
|
+
s0, _ = compute_epistemic_score(base)
|
|
195
|
+
s1, _ = compute_epistemic_score(more_rejects)
|
|
196
|
+
self.assertLessEqual(s1, s0)
|
|
197
|
+
|
|
198
|
+
def test_zero_assertions_clamps(self):
|
|
199
|
+
score, br = compute_epistemic_score(
|
|
200
|
+
{"verified": 0, "rejected": 0, "inferred": 0, "hypotheses": 0, "neg_knowledge": 0}
|
|
201
|
+
)
|
|
202
|
+
self.assertEqual(score, 0.0)
|
|
203
|
+
self.assertEqual(br["total_assertions"], 1)
|
|
204
|
+
|
|
205
|
+
def test_property_loops_200_random_claim_sets(self):
|
|
206
|
+
rng = random.Random(0xF00D)
|
|
207
|
+
for _ in range(200):
|
|
208
|
+
counts = {
|
|
209
|
+
"verified": rng.randint(0, 20),
|
|
210
|
+
"rejected": rng.randint(0, 10),
|
|
211
|
+
"inferred": rng.randint(0, 10),
|
|
212
|
+
"hypotheses": rng.randint(0, 10),
|
|
213
|
+
"neg_knowledge": rng.randint(0, 5),
|
|
214
|
+
}
|
|
215
|
+
score, _br = compute_epistemic_score(counts)
|
|
216
|
+
self.assertGreaterEqual(score, 0.0, counts)
|
|
217
|
+
self.assertLessEqual(score, 1.0, counts)
|
|
218
|
+
# Determinism
|
|
219
|
+
score2, _ = compute_epistemic_score(counts)
|
|
220
|
+
self.assertEqual(score, score2)
|
|
221
|
+
# More rejections never improve the score
|
|
222
|
+
worse = dict(counts, rejected=counts["rejected"] + 1)
|
|
223
|
+
worse_score, _ = compute_epistemic_score(worse)
|
|
224
|
+
self.assertLessEqual(worse_score, score, (counts, worse))
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
class TestConstitution(unittest.TestCase):
|
|
228
|
+
def test_legacy_default_is_flat(self):
|
|
229
|
+
self.assertEqual(LEGACY_CONSTITUTION.weight_for(None), 1.0)
|
|
230
|
+
self.assertEqual(LEGACY_CONSTITUTION.weight_for("PREPRINT"), 1.0)
|
|
231
|
+
|
|
232
|
+
def test_tier_weights_override(self):
|
|
233
|
+
const = Constitution(tier_weights={"PREPRINT": 0.3, "__default__": 1.0})
|
|
234
|
+
self.assertEqual(const.weight_for("PREPRINT"), 0.3)
|
|
235
|
+
self.assertEqual(const.weight_for("PEER_REVIEWED"), 1.0)
|
|
236
|
+
|
|
237
|
+
def test_tier_weighted_claims_path(self):
|
|
238
|
+
# Legacy: 2 verified = 1.0
|
|
239
|
+
legacy_claims = [make_claim(claim_id=f"C{i}") for i in range(2)]
|
|
240
|
+
s_legacy, _ = compute_epistemic_score_from_claims(legacy_claims)
|
|
241
|
+
self.assertEqual(s_legacy, 1.0)
|
|
242
|
+
|
|
243
|
+
# Biopharma-ish: preprints downgraded to 0.3
|
|
244
|
+
const = Constitution(tier_weights={"PREPRINT": 0.3, "__default__": 1.0})
|
|
245
|
+
preprints = [
|
|
246
|
+
make_claim(claim_id=f"C{i}", tier="PREPRINT") for i in range(2)
|
|
247
|
+
]
|
|
248
|
+
s_tiered, br = compute_epistemic_score_from_claims(preprints, constitution=const)
|
|
249
|
+
# weight = 0.3*2 = 0.6, assertions = 2, raw = 0.6/2 = 0.3
|
|
250
|
+
self.assertEqual(s_tiered, 0.3)
|
|
251
|
+
self.assertAlmostEqual(br["verified_weight"], 0.6, places=3)
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
if __name__ == "__main__":
|
|
255
|
+
unittest.main()
|
|
@@ -20,6 +20,7 @@ from runner.state_machine import ResearchStateMachine, SessionStatus, ScopeStatu
|
|
|
20
20
|
from runner.auditor_engine import EpistemicAuditorEngine
|
|
21
21
|
from runner.research_swarm import SwarmRunner
|
|
22
22
|
|
|
23
|
+
|
|
23
24
|
class TestEpistemicSwarm(unittest.TestCase):
|
|
24
25
|
def setUp(self):
|
|
25
26
|
self.test_dir = Path(tempfile.mkdtemp(prefix="epistemic_test_"))
|
|
@@ -38,7 +39,7 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
|
|
|
38
39
|
shash = self.hasher.store_source(
|
|
39
40
|
url="https://arxiv.org/abs/2203.15556",
|
|
40
41
|
content=content,
|
|
41
|
-
title="Chinchilla Scaling Laws"
|
|
42
|
+
title="Chinchilla Scaling Laws",
|
|
42
43
|
)
|
|
43
44
|
self.assertTrue(len(shash) == 64)
|
|
44
45
|
|
|
@@ -57,7 +58,8 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
|
|
|
57
58
|
|
|
58
59
|
# 3. Fabricated quote rejection
|
|
59
60
|
verified_fake, conf_fake, _ = self.hasher.verify_quote(
|
|
60
|
-
shash,
|
|
61
|
+
shash,
|
|
62
|
+
"the 70B parameter model was trained on 500 quadrillion tokens by aliens.",
|
|
61
63
|
)
|
|
62
64
|
self.assertFalse(verified_fake)
|
|
63
65
|
self.assertLess(conf_fake, 0.8)
|
|
@@ -68,13 +70,13 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
|
|
|
68
70
|
{
|
|
69
71
|
"scope_id": "scope_01_prover",
|
|
70
72
|
"title": "Prover Benchmarks",
|
|
71
|
-
"dependencies": []
|
|
73
|
+
"dependencies": [],
|
|
72
74
|
},
|
|
73
75
|
{
|
|
74
76
|
"scope_id": "scope_02_recursion",
|
|
75
77
|
"title": "Recursive Verification",
|
|
76
|
-
"dependencies": ["scope_01_prover"]
|
|
77
|
-
}
|
|
78
|
+
"dependencies": ["scope_01_prover"],
|
|
79
|
+
},
|
|
78
80
|
]
|
|
79
81
|
self.state_machine.set_scopes(scopes)
|
|
80
82
|
|
|
@@ -96,7 +98,7 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
|
|
|
96
98
|
shash = self.hasher.store_source(
|
|
97
99
|
url="https://benchmark.org/zk",
|
|
98
100
|
content="FPGA prover executes Poseidon in 184ms.",
|
|
99
|
-
title="ZK Benchmarks"
|
|
101
|
+
title="ZK Benchmarks",
|
|
100
102
|
)
|
|
101
103
|
|
|
102
104
|
# 2. Initialize scope
|
|
@@ -113,17 +115,17 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
|
|
|
113
115
|
"tag": "VERIFIED",
|
|
114
116
|
"statement": "Poseidon prover executes in 184ms",
|
|
115
117
|
"source_hash": shash,
|
|
116
|
-
"verbatim_quote": "FPGA prover executes Poseidon in 184ms."
|
|
118
|
+
"verbatim_quote": "FPGA prover executes Poseidon in 184ms.",
|
|
117
119
|
},
|
|
118
120
|
{
|
|
119
121
|
"claim_id": "A2",
|
|
120
122
|
"tag": "VERIFIED",
|
|
121
123
|
"statement": "Hallucinated claim that does not exist in source",
|
|
122
124
|
"source_hash": shash,
|
|
123
|
-
"verbatim_quote": "This string does not exist anywhere in the text."
|
|
124
|
-
}
|
|
125
|
+
"verbatim_quote": "This string does not exist anywhere in the text.",
|
|
126
|
+
},
|
|
125
127
|
],
|
|
126
|
-
"negative_knowledge": [{"query": "q1", "finding": "None"}]
|
|
128
|
+
"negative_knowledge": [{"query": "q1", "finding": "None"}],
|
|
127
129
|
}
|
|
128
130
|
|
|
129
131
|
# 4. Create Beta Dossier
|
|
@@ -134,10 +136,10 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
|
|
|
134
136
|
"methodological_critiques": [
|
|
135
137
|
{
|
|
136
138
|
"target_assertion": "Poseidon prover executes in 184ms",
|
|
137
|
-
"critique": "Benchmark excludes PCIe host bus latency"
|
|
139
|
+
"critique": "Benchmark excludes PCIe host bus latency",
|
|
138
140
|
}
|
|
139
141
|
],
|
|
140
|
-
"negative_knowledge": []
|
|
142
|
+
"negative_knowledge": [],
|
|
141
143
|
}
|
|
142
144
|
|
|
143
145
|
self.state_machine.record_agent_completion("scope_test", "alpha", alpha_dossier)
|
|
@@ -189,7 +191,9 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
|
|
|
189
191
|
|
|
190
192
|
for ext_path in pkg["pi"].get("extensions", []):
|
|
191
193
|
resolved = (PROJECT_ROOT / ext_path).resolve()
|
|
192
|
-
self.assertTrue(
|
|
194
|
+
self.assertTrue(
|
|
195
|
+
resolved.exists(), f"Pi extension path not found: {ext_path}"
|
|
196
|
+
)
|
|
193
197
|
|
|
194
198
|
# 2. OpenCode plugin validation
|
|
195
199
|
self.assertIn("opencode", pkg.get("keywords", []))
|
|
@@ -203,6 +207,7 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
|
|
|
203
207
|
|
|
204
208
|
def test_config_manager_load_and_save(self):
|
|
205
209
|
from skills.swarm_config.configure import load_config, save_config
|
|
210
|
+
|
|
206
211
|
cfg = load_config(str(self.test_dir))
|
|
207
212
|
self.assertEqual(cfg["search_engine"], "duckduckgo")
|
|
208
213
|
self.assertEqual(cfg["mode"], "research")
|
|
@@ -237,8 +242,161 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
|
|
|
237
242
|
content = f.read()
|
|
238
243
|
self.assertIn("Open-Source Software Discovery", content)
|
|
239
244
|
|
|
245
|
+
def test_mock_brainstorm_mode(self):
|
|
246
|
+
runner = SwarmRunner(base_dir=self.test_dir, mock_mode=True, mode="brainstorm")
|
|
247
|
+
runner.run_swarm("Where do we go from here?")
|
|
240
248
|
|
|
241
|
-
|
|
242
|
-
|
|
249
|
+
brainstorm_report = self.test_dir / "brainstorm_report.md"
|
|
250
|
+
self.assertTrue(brainstorm_report.exists())
|
|
251
|
+
with open(brainstorm_report, "r") as f:
|
|
252
|
+
content = f.read()
|
|
253
|
+
self.assertIn("Lateral Brainstorm", content)
|
|
254
|
+
|
|
255
|
+
def test_brainstorm_prompt_and_skill_exist(self):
|
|
256
|
+
prompt = PROJECT_ROOT / "prompts" / "agent_brainstormer.md"
|
|
257
|
+
skill = PROJECT_ROOT / "skills" / "brainstorming" / "SKILL.md"
|
|
258
|
+
scaffold = (
|
|
259
|
+
PROJECT_ROOT / "skills" / "brainstorming" / "scripts" / "brainstorm.py"
|
|
260
|
+
)
|
|
261
|
+
self.assertTrue(prompt.exists())
|
|
262
|
+
self.assertTrue(skill.exists())
|
|
263
|
+
self.assertTrue(scaffold.exists())
|
|
264
|
+
text = skill.read_text(encoding="utf-8")
|
|
265
|
+
self.assertIn("name: brainstorming", text)
|
|
266
|
+
self.assertIn("HYPOTHESIS", text)
|
|
267
|
+
|
|
268
|
+
def test_universal_adapter_bundles_exist(self):
|
|
269
|
+
# AntiGravity
|
|
270
|
+
self.assertTrue((PROJECT_ROOT / "plugins/antigravity/plugin.json").exists())
|
|
271
|
+
self.assertTrue((PROJECT_ROOT / "plugins/antigravity/mcp_config.json").exists())
|
|
272
|
+
self.assertTrue((PROJECT_ROOT / "plugins/antigravity/hooks.json").exists())
|
|
273
|
+
self.assertTrue(
|
|
274
|
+
(
|
|
275
|
+
PROJECT_ROOT / "plugins/antigravity/skills/brainstorming/SKILL.md"
|
|
276
|
+
).exists()
|
|
277
|
+
)
|
|
278
|
+
self.assertTrue(
|
|
279
|
+
(PROJECT_ROOT / "plugins/antigravity/agents/brainstormer.md").exists()
|
|
280
|
+
)
|
|
281
|
+
# Gemini
|
|
282
|
+
self.assertTrue(
|
|
283
|
+
(PROJECT_ROOT / "plugins/gemini/gemini-extension.json").exists()
|
|
284
|
+
)
|
|
285
|
+
self.assertTrue((PROJECT_ROOT / "plugins/gemini/GEMINI.md").exists())
|
|
286
|
+
self.assertTrue(
|
|
287
|
+
(PROJECT_ROOT / "plugins/gemini/commands/brainstorming.toml").exists()
|
|
288
|
+
)
|
|
289
|
+
self.assertTrue(
|
|
290
|
+
(PROJECT_ROOT / "plugins/gemini/skills/brainstorming/SKILL.md").exists()
|
|
291
|
+
)
|
|
292
|
+
# Codex
|
|
293
|
+
self.assertTrue((PROJECT_ROOT / "plugins/codex/openai.yaml").exists())
|
|
294
|
+
self.assertTrue((PROJECT_ROOT / "plugins/codex/config.toml.snippet").exists())
|
|
295
|
+
self.assertTrue(
|
|
296
|
+
(PROJECT_ROOT / "plugins/codex/skills/brainstorming/SKILL.md").exists()
|
|
297
|
+
)
|
|
298
|
+
self.assertTrue(
|
|
299
|
+
(PROJECT_ROOT / ".agents/skills/brainstorming/SKILL.md").exists()
|
|
300
|
+
)
|
|
301
|
+
# OMP (oh-my-pi)
|
|
302
|
+
self.assertTrue((PROJECT_ROOT / ".omp/commands/brainstorming.md").exists())
|
|
303
|
+
self.assertTrue((PROJECT_ROOT / ".omp/prompts/brainstorming.md").exists())
|
|
304
|
+
self.assertTrue((PROJECT_ROOT / ".omp/SYSTEM.md").exists())
|
|
305
|
+
self.assertTrue(
|
|
306
|
+
(PROJECT_ROOT / ".omp/hooks/pre/epistemic-redirect.ts").exists()
|
|
307
|
+
)
|
|
308
|
+
# package.json omp block mirrors pi block
|
|
309
|
+
with open(PROJECT_ROOT / "package.json", "r", encoding="utf-8") as f:
|
|
310
|
+
pkg = json.load(f)
|
|
311
|
+
self.assertIn("omp", pkg)
|
|
312
|
+
self.assertIn("./extensions/pi/index.js", pkg["omp"].get("extensions", []))
|
|
313
|
+
self.assertIn("./skills/brainstorming", pkg["omp"].get("skills", []))
|
|
314
|
+
self.assertIn("./skills/brainstorming", pkg["pi"].get("skills", []))
|
|
315
|
+
|
|
316
|
+
def test_opencode_and_pi_extension_interfaces(self):
|
|
317
|
+
import subprocess
|
|
318
|
+
|
|
319
|
+
# 1. Test OpenCode Plugin registration (A2b: thin forwarders + both
|
|
320
|
+
# name exports — canonical IUMBTEMS_TOOL_NAMES and the deprecated
|
|
321
|
+
# IUMBEMS_TOOL_NAMES alias).
|
|
322
|
+
node_code_oc = """
|
|
323
|
+
import plugin, { IUMBTEMS_TOOL_NAMES, IUMBEMS_TOOL_NAMES } from "./plugins/opencode/index.js";
|
|
324
|
+
const tools = plugin.server();
|
|
325
|
+
const names = tools.map(t => t.name);
|
|
326
|
+
console.log(JSON.stringify({
|
|
327
|
+
names,
|
|
328
|
+
canonical: IUMBTEMS_TOOL_NAMES,
|
|
329
|
+
alias: IUMBEMS_TOOL_NAMES
|
|
330
|
+
}));
|
|
331
|
+
"""
|
|
332
|
+
res_oc = subprocess.run(
|
|
333
|
+
["node", "--input-type=module", "-e", node_code_oc],
|
|
334
|
+
capture_output=True,
|
|
335
|
+
text=True,
|
|
336
|
+
cwd=str(PROJECT_ROOT),
|
|
337
|
+
)
|
|
338
|
+
self.assertEqual(
|
|
339
|
+
res_oc.returncode, 0, f"OpenCode plugin test failed: {res_oc.stderr}"
|
|
340
|
+
)
|
|
341
|
+
lines = [
|
|
342
|
+
line.strip()
|
|
343
|
+
for line in res_oc.stdout.strip().split("\n")
|
|
344
|
+
if line.strip().startswith("{")
|
|
345
|
+
]
|
|
346
|
+
data_oc = json.loads(lines[-1])
|
|
347
|
+
tools = data_oc["names"]
|
|
348
|
+
# Canonical + deprecated alias must both exist and agree.
|
|
349
|
+
self.assertEqual(data_oc["canonical"], data_oc["alias"])
|
|
350
|
+
self.assertEqual(sorted(data_oc["canonical"]), sorted(tools))
|
|
351
|
+
expected_tools = [
|
|
352
|
+
"iumbtems_config",
|
|
353
|
+
"iumbtems_swarm_research",
|
|
354
|
+
"iumbtems_code_audit",
|
|
355
|
+
"iumbtems_oss_scout",
|
|
356
|
+
"iumbtems_brainstorm",
|
|
357
|
+
"iumbtems_verify_quote",
|
|
358
|
+
"iumbtems_socratic_frontier",
|
|
359
|
+
"iumbtems_reindex_claims",
|
|
360
|
+
"iumbtems_report_retraction",
|
|
361
|
+
"iumbtems_check_staleness",
|
|
362
|
+
"iumbtems_set_domain_pack",
|
|
363
|
+
"iumbtems_export_brief",
|
|
364
|
+
"iumbtems_verify_brief",
|
|
365
|
+
]
|
|
366
|
+
for t in expected_tools:
|
|
367
|
+
self.assertIn(t, tools)
|
|
368
|
+
|
|
369
|
+
# 2. Test Pi Extension registration
|
|
370
|
+
node_code_pi = """
|
|
371
|
+
import initPi from "./extensions/pi/index.js";
|
|
372
|
+
const commands = [];
|
|
373
|
+
const tools = [];
|
|
374
|
+
initPi({
|
|
375
|
+
registerCommand: (name) => commands.push(name),
|
|
376
|
+
registerTool: (def) => tools.push(def.name)
|
|
377
|
+
});
|
|
378
|
+
console.log(JSON.stringify({ commands, tools }));
|
|
379
|
+
"""
|
|
380
|
+
res_pi = subprocess.run(
|
|
381
|
+
["node", "--input-type=module", "-e", node_code_pi],
|
|
382
|
+
capture_output=True,
|
|
383
|
+
text=True,
|
|
384
|
+
cwd=str(PROJECT_ROOT),
|
|
385
|
+
)
|
|
386
|
+
self.assertEqual(
|
|
387
|
+
res_pi.returncode, 0, f"Pi extension test failed: {res_pi.stderr}"
|
|
388
|
+
)
|
|
389
|
+
lines = [
|
|
390
|
+
line.strip()
|
|
391
|
+
for line in res_pi.stdout.strip().split("\n")
|
|
392
|
+
if line.strip().startswith("{")
|
|
393
|
+
]
|
|
394
|
+
data = json.loads(lines[-1])
|
|
395
|
+
for c in ["swarm", "grill", "swarm-config", "audit", "scout", "brainstorming"]:
|
|
396
|
+
self.assertIn(c, data["commands"])
|
|
397
|
+
for t in ["iumbtems_verify_quote", "iumbtems_config", "iumbtems_brainstorm"]:
|
|
398
|
+
self.assertIn(t, data["tools"])
|
|
243
399
|
|
|
244
400
|
|
|
401
|
+
if __name__ == "__main__":
|
|
402
|
+
unittest.main()
|