@heretek-ai/epistemic-swarm 0.2.2 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (138) hide show
  1. package/.agents/skills/brainstorming/SKILL.md +13 -0
  2. package/.agents/skills/code_audit/SKILL.md +13 -0
  3. package/.agents/skills/epistemic_search/SKILL.md +13 -0
  4. package/.agents/skills/grilling/SKILL.md +13 -0
  5. package/.agents/skills/oss_scout/SKILL.md +13 -0
  6. package/.agents/skills/research_cache/SKILL.md +13 -0
  7. package/.agents/skills/swarm_config/SKILL.md +13 -0
  8. package/.claude-plugin/plugin.json +15 -5
  9. package/.omp/README.md +39 -0
  10. package/.omp/SYSTEM.md +12 -0
  11. package/.omp/commands/audit.md +10 -0
  12. package/.omp/commands/brainstorming.md +12 -0
  13. package/.omp/commands/grill.md +9 -0
  14. package/.omp/commands/scout.md +10 -0
  15. package/.omp/commands/swarm-config.md +10 -0
  16. package/.omp/commands/swarm.md +9 -0
  17. package/.omp/hooks/post/epistemic-audit.ts +16 -0
  18. package/.omp/hooks/pre/epistemic-redirect.ts +21 -0
  19. package/.omp/prompts/brainstorming.md +8 -0
  20. package/.omp/prompts/swarm.md +8 -0
  21. package/MARKETPLACE.md +8 -0
  22. package/README.md +53 -8
  23. package/bin/cli.js +27 -3
  24. package/config/domain_packs/biopharma.json +23 -0
  25. package/config/domain_packs/legal.json +19 -0
  26. package/config/domain_packs/quant.json +19 -0
  27. package/config/mcp-research-servers.json +7 -0
  28. package/config/mcp_launcher.py +48 -137
  29. package/config/opencode-snippet.json +58 -3
  30. package/config/searxng_mcp.py +42 -83
  31. package/extensions/pi/index.js +196 -28
  32. package/install.sh +20 -4
  33. package/package.json +40 -5
  34. package/plugins/antigravity/README.md +28 -0
  35. package/plugins/antigravity/agents/alpha-thesis.md +6 -0
  36. package/plugins/antigravity/agents/beta-antithesis.md +7 -0
  37. package/plugins/antigravity/agents/brainstormer.md +7 -0
  38. package/plugins/antigravity/agents/epistemic-auditor.md +5 -0
  39. package/plugins/antigravity/hooks.json +23 -0
  40. package/plugins/antigravity/mcp_config.json +33 -0
  41. package/plugins/antigravity/plugin.json +21 -0
  42. package/plugins/antigravity/rules/epistemic-integrity.md +6 -0
  43. package/plugins/antigravity/skills/brainstorming/SKILL.md +13 -0
  44. package/plugins/antigravity/skills/code_audit/SKILL.md +13 -0
  45. package/plugins/antigravity/skills/epistemic_search/SKILL.md +13 -0
  46. package/plugins/antigravity/skills/grilling/SKILL.md +13 -0
  47. package/plugins/antigravity/skills/oss_scout/SKILL.md +13 -0
  48. package/plugins/antigravity/skills/research_cache/SKILL.md +13 -0
  49. package/plugins/antigravity/skills/swarm_config/SKILL.md +13 -0
  50. package/plugins/codex/AGENTS.md.snippet +10 -0
  51. package/plugins/codex/README.md +37 -0
  52. package/plugins/codex/config.toml.snippet +28 -0
  53. package/plugins/codex/openai.yaml +24 -0
  54. package/plugins/codex/skills/brainstorming/SKILL.md +13 -0
  55. package/plugins/codex/skills/code_audit/SKILL.md +13 -0
  56. package/plugins/codex/skills/epistemic_search/SKILL.md +13 -0
  57. package/plugins/codex/skills/grilling/SKILL.md +13 -0
  58. package/plugins/codex/skills/oss_scout/SKILL.md +13 -0
  59. package/plugins/codex/skills/research_cache/SKILL.md +13 -0
  60. package/plugins/codex/skills/swarm_config/SKILL.md +13 -0
  61. package/plugins/gemini/GEMINI.md +15 -0
  62. package/plugins/gemini/README.md +19 -0
  63. package/plugins/gemini/commands/audit.toml +6 -0
  64. package/plugins/gemini/commands/brainstorming.toml +10 -0
  65. package/plugins/gemini/commands/grill.toml +6 -0
  66. package/plugins/gemini/commands/scout.toml +7 -0
  67. package/plugins/gemini/commands/swarm-config.toml +7 -0
  68. package/plugins/gemini/commands/swarm.toml +8 -0
  69. package/plugins/gemini/gemini-extension.json +38 -0
  70. package/plugins/gemini/hooks/hooks.json +11 -0
  71. package/plugins/gemini/skills/brainstorming/SKILL.md +13 -0
  72. package/plugins/gemini/skills/code_audit/SKILL.md +13 -0
  73. package/plugins/gemini/skills/epistemic_search/SKILL.md +13 -0
  74. package/plugins/gemini/skills/grilling/SKILL.md +13 -0
  75. package/plugins/gemini/skills/oss_scout/SKILL.md +13 -0
  76. package/plugins/gemini/skills/research_cache/SKILL.md +13 -0
  77. package/plugins/gemini/skills/swarm_config/SKILL.md +13 -0
  78. package/plugins/opencode/index.js +335 -118
  79. package/prompts/agent_brainstormer.md +97 -0
  80. package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
  81. package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
  82. package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
  83. package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
  84. package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
  85. package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
  86. package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
  87. package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
  88. package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
  89. package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
  90. package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
  91. package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
  92. package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
  93. package/runner/auctioneer.py +169 -0
  94. package/runner/auditor_engine.py +127 -12
  95. package/runner/claim_store.py +361 -0
  96. package/runner/claim_witness.py +183 -0
  97. package/runner/living_dossiers.py +355 -0
  98. package/runner/mcp_protocol.py +188 -0
  99. package/runner/mcp_server.py +567 -0
  100. package/runner/pcrb.py +212 -0
  101. package/runner/pcrb_verify.py +237 -0
  102. package/runner/refinement.py +335 -0
  103. package/runner/research_swarm.py +378 -94
  104. package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
  105. package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
  106. package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
  107. package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
  108. package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
  109. package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
  110. package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
  111. package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
  112. package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
  113. package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
  114. package/runner/tests/fixtures/auction_objective.json +35 -0
  115. package/runner/tests/fixtures/borderline_claims.json +27 -0
  116. package/runner/tests/fixtures/divergence_objectives.json +31 -0
  117. package/runner/tests/test_auction_order.py +173 -0
  118. package/runner/tests/test_claim_store.py +162 -0
  119. package/runner/tests/test_claim_witness.py +186 -0
  120. package/runner/tests/test_domain_packs.py +217 -0
  121. package/runner/tests/test_fleet_seam.py +113 -0
  122. package/runner/tests/test_living_dossiers.py +204 -0
  123. package/runner/tests/test_mcp_server.py +212 -0
  124. package/runner/tests/test_pcrb.py +240 -0
  125. package/runner/tests/test_refinement.py +255 -0
  126. package/runner/tests/test_swarm.py +173 -15
  127. package/scripts/auction_experiment.py +180 -0
  128. package/scripts/build_adapters.py +183 -0
  129. package/scripts/divergence_experiment.py +184 -0
  130. package/skills/brainstorming/SKILL.md +106 -0
  131. package/skills/brainstorming/__init__.py +1 -0
  132. package/skills/brainstorming/scripts/brainstorm.py +200 -0
  133. package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
  134. package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
  135. package/skills/swarm_config/SKILL.md +1 -1
  136. package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
  137. package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
  138. package/skills/swarm_config/configure.py +45 -11
@@ -0,0 +1,35 @@
1
+ {
2
+ "objective_set": "auction-f-v1",
3
+ "note": "Fixed 4-scope objective for the Stream F auction-vs-DAG probe. Stable set: do not edit between runs, or wall-clock and claims/token are not comparable.",
4
+ "objective": "Evaluate the operational envelope of content-addressed research caches for long-lived engineering briefs",
5
+ "scopes": [
6
+ {
7
+ "scope_id": "scope_01_integrity",
8
+ "title": "Tamper evidence in content-addressed caches",
9
+ "objective": "Assess mechanisms for detecting tampering in SHA-256 addressed document caches",
10
+ "dependencies": [],
11
+ "bid": 4.0
12
+ },
13
+ {
14
+ "scope_id": "scope_02_staleness",
15
+ "title": "Claim staleness and source retraction",
16
+ "objective": "Determine how retraction and revision events should degrade previously verified claims",
17
+ "dependencies": ["scope_01_integrity"],
18
+ "bid": 3.0
19
+ },
20
+ {
21
+ "scope_id": "scope_03_portability",
22
+ "title": "Portable proof-carrying briefs",
23
+ "objective": "Evaluate self-contained signed brief bundles for third-party re-verification",
24
+ "dependencies": [],
25
+ "bid": 2.0
26
+ },
27
+ {
28
+ "scope_id": "scope_04_allocation",
29
+ "title": "Research resource allocation policies",
30
+ "objective": "Compare static DAG ordering versus bid-based allocation of dialectic research scopes",
31
+ "dependencies": [],
32
+ "bid": 1.0
33
+ }
34
+ ]
35
+ }
@@ -0,0 +1,27 @@
1
+ {
2
+ "fixture_id": "borderline-g-v1",
3
+ "note": "20 borderline claims for the Domain Packs probe. Mechanisms: 6 banned-domain, 2 missing-tag, 2 retracted-source (see retracted_hashes), 10 clean preprint-tier. Default constitution should ACCEPT all 20; biopharma should REJECT >= 25%.",
4
+ "retracted_hashes": ["RETRACTED_SRC_A", "RETRACTED_SRC_B"],
5
+ "claims": [
6
+ {"claim_id": "BB-01", "tag": "VERIFIED", "statement": "Cohort endpoint met primary efficacy bar", "source_hash": "PREPRINT_01", "source_url": "https://banned-preprint-mirror.example/p1", "verbatim_quote": "primary efficacy bar met", "tier": "PREPRINT"},
7
+ {"claim_id": "BB-02", "tag": "VERIFIED", "statement": "Adverse event rate within expected band", "source_hash": "PREPRINT_02", "source_url": "https://banned-preprint-mirror.example/p2", "verbatim_quote": "within expected band", "tier": "PREPRINT"},
8
+ {"claim_id": "BB-03", "tag": "VERIFIED", "statement": "Secondary endpoint shows dose response", "source_hash": "PREPRINT_03", "source_url": "https://banned-preprint-mirror.example/p3", "verbatim_quote": "dose response observed", "tier": "PREPRINT"},
9
+ {"claim_id": "BB-04", "tag": "VERIFIED", "statement": "Biomarker subgroup analysis favorable", "source_hash": "PREPRINT_04", "source_url": "https://predatory-journal.example/p4", "verbatim_quote": "subgroup analysis favorable", "tier": "PREPRINT"},
10
+ {"claim_id": "BB-05", "tag": "VERIFIED", "statement": "Pharmacokinetics consistent with model", "source_hash": "PREPRINT_05", "source_url": "https://predatory-journal.example/p5", "verbatim_quote": "consistent with model", "tier": "PREPRINT"},
11
+ {"claim_id": "BB-06", "tag": "VERIFIED", "statement": "Manufacturing yield acceptable at scale", "source_hash": "PREPRINT_06", "source_url": "https://banned-preprint-mirror.example/p6", "verbatim_quote": "yield acceptable at scale", "tier": "PREPRINT"},
12
+ {"claim_id": "BB-07", "tag": "", "statement": "Untagged assertion lacking epistemic classification", "source_hash": "CLEAN_01", "source_url": "https://journal.example/c1", "verbatim_quote": "assertion recorded without tag", "tier": "PREPRINT"},
13
+ {"claim_id": "BB-08", "tag": "", "statement": "Second untagged assertion in the borderline set", "source_hash": "CLEAN_02", "source_url": "https://journal.example/c2", "verbatim_quote": "second assertion without tag", "tier": "PREPRINT"},
14
+ {"claim_id": "BB-09", "tag": "VERIFIED", "statement": "Claim resting on a retracted source", "source_hash": "RETRACTED_SRC_A", "source_url": "https://journal.example/r1", "verbatim_quote": "finding later retracted", "tier": "PREPRINT"},
15
+ {"claim_id": "BB-10", "tag": "VERIFIED", "statement": "Second claim resting on a retracted source", "source_hash": "RETRACTED_SRC_B", "source_url": "https://journal.example/r2", "verbatim_quote": "second retracted finding", "tier": "PREPRINT"},
16
+ {"claim_id": "BB-11", "tag": "VERIFIED", "statement": "Preprint claim near acceptance boundary", "source_hash": "CLEAN_03", "source_url": "https://journal.example/c3", "verbatim_quote": "near acceptance boundary", "tier": "PREPRINT"},
17
+ {"claim_id": "BB-12", "tag": "VERIFIED", "statement": "Preprint claim with modest evidence", "source_hash": "CLEAN_04", "source_url": "https://journal.example/c4", "verbatim_quote": "modest evidence presented", "tier": "PREPRINT"},
18
+ {"claim_id": "BB-13", "tag": "VERIFIED", "statement": "Preprint claim with weak statistical power", "source_hash": "CLEAN_05", "source_url": "https://journal.example/c5", "verbatim_quote": "weak statistical power", "tier": "PREPRINT"},
19
+ {"claim_id": "BB-14", "tag": "VERIFIED", "statement": "Preprint claim awaiting replication", "source_hash": "CLEAN_06", "source_url": "https://journal.example/c6", "verbatim_quote": "awaiting replication", "tier": "PREPRINT"},
20
+ {"claim_id": "BB-15", "tag": "VERIFIED", "statement": "Preprint claim with narrow applicability", "source_hash": "CLEAN_07", "source_url": "https://journal.example/c7", "verbatim_quote": "narrow applicability", "tier": "PREPRINT"},
21
+ {"claim_id": "BB-16", "tag": "VERIFIED", "statement": "Preprint claim with exploratory endpoint", "source_hash": "CLEAN_08", "source_url": "https://journal.example/c8", "verbatim_quote": "exploratory endpoint", "tier": "PREPRINT"},
22
+ {"claim_id": "BB-17", "tag": "VERIFIED", "statement": "Preprint claim with post-hoc analysis", "source_hash": "CLEAN_09", "source_url": "https://journal.example/c9", "verbatim_quote": "post-hoc analysis", "tier": "PREPRINT"},
23
+ {"claim_id": "BB-18", "tag": "VERIFIED", "statement": "Preprint claim with surrogate endpoint", "source_hash": "CLEAN_10", "source_url": "https://journal.example/c10", "verbatim_quote": "surrogate endpoint", "tier": "PREPRINT"},
24
+ {"claim_id": "BB-19", "tag": "VERIFIED", "statement": "Preprint claim with single-site data", "source_hash": "CLEAN_11", "source_url": "https://journal.example/c11", "verbatim_quote": "single-site data", "tier": "PREPRINT"},
25
+ {"claim_id": "BB-20", "tag": "VERIFIED", "statement": "Preprint claim with interim readout", "source_hash": "CLEAN_12", "source_url": "https://journal.example/c12", "verbatim_quote": "interim readout", "tier": "PREPRINT"}
26
+ ]
27
+ }
@@ -0,0 +1,31 @@
1
+ {
2
+ "objective_set": "divergence-g2-v1",
3
+ "note": "5 fixed objectives for the Stream E cross-family divergence probe. Stable set: do not edit between runs, or the delta is not comparable.",
4
+ "objectives": [
5
+ {
6
+ "id": "G2-01",
7
+ "objective": "Evaluate feasibility of sub-100ms Poseidon witness generation on commodity GPUs",
8
+ "domain": "zk-prover-hardware"
9
+ },
10
+ {
11
+ "id": "G2-02",
12
+ "objective": "Assess whether content-addressed citation caches can be made tamper-evident without a PKI",
13
+ "domain": "epistemic-infrastructure"
14
+ },
15
+ {
16
+ "id": "G2-03",
17
+ "objective": "Determine the conditions under which multi-agent dialectic research outperforms single-model deep research",
18
+ "domain": "multi-agent-methodology"
19
+ },
20
+ {
21
+ "id": "G2-04",
22
+ "objective": "Evaluate zero-dependency stdio MCP servers versus SDK-based MCP servers for CLI-distributed tooling",
23
+ "domain": "developer-tooling"
24
+ },
25
+ {
26
+ "id": "G2-05",
27
+ "objective": "Assess whether claim-level staleness tracking changes the reliability of long-lived research briefs",
28
+ "domain": "research-durability"
29
+ }
30
+ ]
31
+ }
@@ -0,0 +1,173 @@
1
+ #!/usr/bin/env python3
2
+ """Tests for the Stream F Frontier Markets auctioneer (deterministic, mock).
3
+
4
+ The live probe (wall-clock parity + >=20% claims/token) is scripts/auction_experiment.py
5
+ — it needs real models and is NOT a CI step. These tests pin the allocation
6
+ semantics: bid ordering, explicit-bid override, DAG default untouched.
7
+ """
8
+
9
+ import json
10
+ import sys
11
+ import tempfile
12
+ import unittest
13
+ from pathlib import Path
14
+
15
+ PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
16
+ if str(PROJECT_ROOT) not in sys.path:
17
+ sys.path.insert(0, str(PROJECT_ROOT))
18
+
19
+ from runner.auctioneer import ( # noqa: E402
20
+ bid_scope,
21
+ dependency_depth,
22
+ estimate_tokens,
23
+ order_scope_ids,
24
+ pick_next,
25
+ record_scope_telemetry,
26
+ score_scopes,
27
+ unresolved_frontier_questions,
28
+ )
29
+
30
+
31
+ def scope(sid, deps=None, bid=None):
32
+ s = {"scope_id": sid, "dependencies": deps or []}
33
+ if bid is not None:
34
+ s["bid"] = bid
35
+ return s
36
+
37
+
38
+ class TestBidHeuristic(unittest.TestCase):
39
+ def test_dependency_depth(self):
40
+ scopes = [
41
+ scope("a"),
42
+ scope("b", ["a"]),
43
+ scope("c", ["b"]),
44
+ scope("d"),
45
+ ]
46
+ self.assertEqual(dependency_depth(scopes[0], scopes), 0)
47
+ self.assertEqual(dependency_depth(scopes[1], scopes), 1)
48
+ self.assertEqual(dependency_depth(scopes[2], scopes), 2)
49
+ self.assertEqual(dependency_depth(scopes[3], scopes), 0)
50
+
51
+ def test_unresolved_frontier_questions(self):
52
+ frontier = {
53
+ "nodes": {
54
+ "n1": {"question": "q1", "settled_answer": "yes"},
55
+ "n2": {"question": "q2", "settled_answer": None},
56
+ "n3": {"question": "q3"},
57
+ }
58
+ }
59
+ self.assertEqual(unresolved_frontier_questions(frontier), 2)
60
+ self.assertEqual(unresolved_frontier_questions({}), 0)
61
+
62
+ def test_explicit_bid_wins(self):
63
+ s = scope("a", bid=42.5)
64
+ self.assertEqual(bid_scope(s), 42.5)
65
+
66
+ def test_frontier_weighting_prefers_shallow(self):
67
+ frontier = {"nodes": {"n1": {"question": "q", "settled_answer": None}}}
68
+ shallow = scope("a")
69
+ deep = scope("c", ["b"])
70
+ b_shallow = bid_scope(shallow, all_scopes=[shallow, scope("b"), deep], frontier=frontier)
71
+ b_deep = bid_scope(deep, all_scopes=[shallow, scope("b"), deep], frontier=frontier)
72
+ self.assertGreater(b_shallow, b_deep)
73
+
74
+ def test_deterministic(self):
75
+ scopes = [scope("b", ["a"]), scope("a"), scope("c", ["a"], bid=0.1)]
76
+ frontier = {"nodes": {"n1": {"settled_answer": None}}}
77
+ first = [b for b, _s in score_scopes(scopes, frontier=frontier)]
78
+ second = [b for b, _s in score_scopes(scopes, frontier=frontier)]
79
+ self.assertEqual(first, second)
80
+
81
+
82
+ class TestAuctionOrdering(unittest.TestCase):
83
+ def test_orders_by_descending_bid(self):
84
+ scopes = [
85
+ scope("low", bid=0.1),
86
+ scope("high", bid=9.0),
87
+ scope("mid", bid=5.0),
88
+ ]
89
+ self.assertEqual(order_scope_ids(scopes), ["high", "mid", "low"])
90
+
91
+ def test_tie_breaks_on_scope_id(self):
92
+ scopes = [scope("zzz", bid=1.0), scope("aaa", bid=1.0)]
93
+ self.assertEqual(order_scope_ids(scopes), ["aaa", "zzz"])
94
+
95
+ def test_pick_next_respects_budget(self):
96
+ scopes = [scope("high", bid=9.0), scope("mid", bid=5.0), scope("low", bid=0.5)]
97
+ self.assertEqual(pick_next(scopes)["scope_id"], "high")
98
+ self.assertEqual(pick_next(scopes, budget=6.0)["scope_id"], "mid")
99
+ self.assertIsNone(pick_next(scopes, budget=0.1))
100
+
101
+ def test_ready_batch_reordering(self):
102
+ """The exact shape run_swarm uses: a ready batch reordered by bid."""
103
+ ready = [scope("scope_02", ["scope_01"], bid=0.5), scope("scope_03", bid=3.0), scope("scope_01", bid=1.0)]
104
+ ordered = order_scope_ids(ready)
105
+ self.assertEqual(ordered, ["scope_03", "scope_01", "scope_02"])
106
+
107
+
108
+ class TestTelemetry(unittest.TestCase):
109
+ def test_record_scope_telemetry(self):
110
+ with tempfile.TemporaryDirectory() as tmp:
111
+ base = Path(tmp)
112
+ smdir = base / "scratchpads" / "scope_01"
113
+ smdir.mkdir(parents=True)
114
+ (smdir / "manifest.json").write_text(json.dumps({"scope_id": "scope_01", "status": "COMPLETE"}))
115
+
116
+ record_scope_telemetry(base, "scope_01", tokens_used=1234, verified_claims=7, bid=2.5)
117
+ manifest = json.loads((smdir / "manifest.json").read_text())
118
+ self.assertEqual(manifest["telemetry"]["tokens_used"], 1234)
119
+ self.assertEqual(manifest["telemetry"]["verified_claims"], 7)
120
+ self.assertEqual(manifest["telemetry"]["bid"], 2.5)
121
+
122
+ def test_estimate_tokens_flagged_approximation(self):
123
+ self.assertEqual(estimate_tokens("x" * 400), 100)
124
+ self.assertGreaterEqual(estimate_tokens(""), 1)
125
+
126
+
127
+ class TestSwarmAuctionIntegration(unittest.TestCase):
128
+ def test_dag_default_unchanged(self):
129
+ from runner.research_swarm import SwarmRunner
130
+
131
+ with tempfile.TemporaryDirectory() as tmp:
132
+ runner = SwarmRunner(base_dir=Path(tmp), mock_mode=True)
133
+ self.assertEqual(runner.allocation, "dag")
134
+
135
+ def test_auction_config_selected(self):
136
+ from runner.research_swarm import SwarmRunner
137
+
138
+ with tempfile.TemporaryDirectory() as tmp:
139
+ base = Path(tmp)
140
+ (base / "config.json").write_text(json.dumps({"allocation": "auction"}))
141
+ runner = SwarmRunner(base_dir=base, mock_mode=True)
142
+ self.assertEqual(runner.allocation, "auction")
143
+
144
+ def test_cli_override_beats_config(self):
145
+ from runner.research_swarm import SwarmRunner
146
+
147
+ with tempfile.TemporaryDirectory() as tmp:
148
+ base = Path(tmp)
149
+ (base / "config.json").write_text(json.dumps({"allocation": "dag"}))
150
+ runner = SwarmRunner(base_dir=base, mock_mode=True, allocation="auction")
151
+ self.assertEqual(runner.allocation, "auction")
152
+
153
+ def test_mock_swarm_runs_under_auction(self):
154
+ """End-to-end mock run with auction allocation completes and records telemetry."""
155
+ from runner.research_swarm import SwarmRunner
156
+
157
+ with tempfile.TemporaryDirectory() as tmp:
158
+ base = Path(tmp)
159
+ runner = SwarmRunner(base_dir=base, mock_mode=True, allocation="auction")
160
+ runner.run_swarm("Evaluate auction allocation ordering")
161
+
162
+ manifest = runner.state_machine.load_global_manifest()
163
+ self.assertEqual(manifest["status"], "COMPLETED")
164
+ # Mock orchestrator emits one scope; its manifest must carry telemetry.
165
+ sid = manifest["scopes"][0]["scope_id"]
166
+ sm = json.loads((base / "scratchpads" / sid / "manifest.json").read_text())
167
+ self.assertIn("telemetry", sm)
168
+ self.assertIn("tokens_used", sm["telemetry"])
169
+ self.assertIn("verified_claims", sm["telemetry"])
170
+
171
+
172
+ if __name__ == "__main__":
173
+ unittest.main()
@@ -0,0 +1,162 @@
1
+ #!/usr/bin/env python3
2
+ """Tests for the derived claim index (runner/claim_store.py).
3
+
4
+ Invariant under test: the flat files are source of truth; claims.sqlite is
5
+ rebuildable and idempotent. Double-reindex must yield an identical content
6
+ fingerprint; cross-run dedup must collapse shared source hashes.
7
+ """
8
+
9
+ import json
10
+ import sys
11
+ import tempfile
12
+ import unittest
13
+ from pathlib import Path
14
+
15
+ PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
16
+ if str(PROJECT_ROOT) not in sys.path:
17
+ sys.path.insert(0, str(PROJECT_ROOT))
18
+
19
+ from runner.claim_store import ClaimStore, reindex # noqa: E402
20
+
21
+
22
+ def write_dossier(base: Path, scope: str, name: str, payload: dict) -> Path:
23
+ d = base / "scratchpads" / scope
24
+ d.mkdir(parents=True, exist_ok=True)
25
+ p = d / name
26
+ p.write_text(json.dumps(payload, indent=2), encoding="utf-8")
27
+ return p
28
+
29
+
30
+ class TestClaimStore(unittest.TestCase):
31
+ def test_reindex_twice_identical_fingerprint(self):
32
+ with tempfile.TemporaryDirectory() as tmp:
33
+ base = Path(tmp)
34
+ write_dossier(
35
+ base,
36
+ "scope_01",
37
+ "alpha_dossier.json",
38
+ {
39
+ "affirmative_claims": [
40
+ {"claim_id": "A1", "statement": "s", "source_hash": "h1", "verbatim_quote": "q"}
41
+ ],
42
+ "negative_knowledge": [{"query": "q", "finding": "f"}],
43
+ },
44
+ )
45
+ write_dossier(
46
+ base,
47
+ "scope_01",
48
+ "beta_dossier.json",
49
+ {
50
+ "falsification_claims": [
51
+ {"claim_id": "B1", "statement": "t", "source_hash": "h1", "verbatim_quote": "q2", "severity": "LOW"}
52
+ ],
53
+ },
54
+ )
55
+
56
+ r1 = reindex(base)
57
+ r2 = reindex(base)
58
+ self.assertEqual(r1["fingerprint"], r2["fingerprint"], "reindex must be idempotent")
59
+ # claims_from_dossier folds all four shapes, so negative_knowledge
60
+ # rows also become ClaimWitness records: 1 affirmative + 1 neg_knowledge
61
+ # + 1 falsification = 3.
62
+ self.assertEqual(r1["claims_indexed"], 3)
63
+ self.assertEqual(r2["claims_indexed"], 3)
64
+
65
+ def test_reindex_captures_sources(self):
66
+ with tempfile.TemporaryDirectory() as tmp:
67
+ base = Path(tmp)
68
+ (base / "sources").mkdir(parents=True)
69
+ (base / "sources" / "h1.json").write_text(
70
+ json.dumps({"hash": "h1", "url": "https://x", "title": "T", "tier": "WEB_DOCUMENT"})
71
+ )
72
+ (base / "sources" / "h1.md").write_text("content")
73
+
74
+ result = reindex(base)
75
+ self.assertEqual(result["sources_indexed"], 1)
76
+
77
+ store = ClaimStore(base)
78
+ rows = store.conn.execute("SELECT * FROM sources").fetchall()
79
+ self.assertEqual(len(rows), 1)
80
+ self.assertEqual(rows[0]["url"], "https://x")
81
+ store.close()
82
+
83
+ def test_cross_run_dedup(self):
84
+ """Two runs citing one source hash => one sources row, N claim rows."""
85
+ with tempfile.TemporaryDirectory() as tmp:
86
+ base = Path(tmp)
87
+ # Register the shared source once (as a cache metadata file).
88
+ (base / "sources").mkdir(parents=True)
89
+ (base / "sources" / "SHARED_HASH.json").write_text(
90
+ json.dumps({"hash": "SHARED_HASH", "url": "https://shared", "title": "Shared"})
91
+ )
92
+ (base / "sources" / "SHARED_HASH.md").write_text("body")
93
+
94
+ for scope in ("scope_01", "scope_02"):
95
+ write_dossier(
96
+ base,
97
+ scope,
98
+ "alpha_dossier.json",
99
+ {
100
+ "affirmative_claims": [
101
+ {
102
+ "claim_id": f"{scope}-A1",
103
+ "statement": f"claim from {scope}",
104
+ "source_hash": "SHARED_HASH",
105
+ "verbatim_quote": "q",
106
+ }
107
+ ]
108
+ },
109
+ )
110
+ reindex(base)
111
+
112
+ store = ClaimStore(base)
113
+ claims = store.claims_citing("SHARED_HASH")
114
+ self.assertEqual(len(claims), 2, "both claims present")
115
+ # One source row (deduped by primary key)
116
+ n_sources = store.conn.execute(
117
+ "SELECT COUNT(*) AS n FROM sources WHERE source_hash = ?", ("SHARED_HASH",)
118
+ ).fetchone()["n"]
119
+ self.assertEqual(n_sources, 1)
120
+ store.close()
121
+
122
+ def test_search_statements(self):
123
+ with tempfile.TemporaryDirectory() as tmp:
124
+ base = Path(tmp)
125
+ write_dossier(
126
+ base,
127
+ "scope_01",
128
+ "alpha_dossier.json",
129
+ {
130
+ "affirmative_claims": [
131
+ {"claim_id": "A1", "statement": "Poseidon hash prover latency", "source_hash": "h", "verbatim_quote": "q"}
132
+ ]
133
+ },
134
+ )
135
+ reindex(base)
136
+ store = ClaimStore(base)
137
+ hits = store.search_statements("Poseidon")
138
+ self.assertEqual(len(hits), 1)
139
+ self.assertEqual(hits[0]["claim_id"], "A1")
140
+ store.close()
141
+
142
+ def test_status_event_append(self):
143
+ with tempfile.TemporaryDirectory() as tmp:
144
+ base = Path(tmp)
145
+ store = ClaimStore(base)
146
+ store.record_status_event("scope_01", "A1", "LIVE", "STALE", "retracted", "2026-09-26T00:00:00Z")
147
+ store.record_status_event("scope_01", "A1", "STALE", "SUSPECT", "revised", "2026-09-26T01:00:00Z")
148
+ rows = store.conn.execute("SELECT * FROM status_events ORDER BY event_id").fetchall()
149
+ self.assertEqual(len(rows), 2)
150
+ self.assertEqual(rows[0]["to_status"], "STALE")
151
+ self.assertEqual(rows[1]["to_status"], "SUSPECT")
152
+ store.close()
153
+
154
+ def test_fts5_flag_is_boolean(self):
155
+ with tempfile.TemporaryDirectory() as tmp:
156
+ store = ClaimStore(Path(tmp))
157
+ self.assertIsInstance(store.fts5, bool)
158
+ store.close()
159
+
160
+
161
+ if __name__ == "__main__":
162
+ unittest.main()
@@ -0,0 +1,186 @@
1
+ #!/usr/bin/env python3
2
+ """Tests for the normalized ClaimWitness model (runner/claim_witness.py)."""
3
+
4
+ import sys
5
+ import unittest
6
+ from pathlib import Path
7
+
8
+ PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
9
+ if str(PROJECT_ROOT) not in sys.path:
10
+ sys.path.insert(0, str(PROJECT_ROOT))
11
+
12
+ from runner.claim_witness import ( # noqa: E402
13
+ KIND_CLAIM,
14
+ KIND_HYPOTHESIS,
15
+ KIND_INFERENCE,
16
+ KIND_NEGATIVE_KNOWLEDGE,
17
+ STATUS_LIVE,
18
+ TAG_HYPOTHESIS,
19
+ TAG_INFERRED,
20
+ TAG_NEGATIVE_KNOWLEDGE,
21
+ TAG_VERIFIED,
22
+ claims_from_dossier,
23
+ normalize_claim,
24
+ witness_check,
25
+ )
26
+
27
+
28
+ class TestNormalizeClaim(unittest.TestCase):
29
+ def test_normalize_claim_kind_claim(self):
30
+ raw = {
31
+ "claim_id": "C-01",
32
+ "tag": "VERIFIED",
33
+ "statement": "s",
34
+ "source_hash": "abc",
35
+ "source_url": "u",
36
+ "verbatim_quote": "q",
37
+ "severity": "LOW",
38
+ }
39
+ c = normalize_claim(raw, kind=KIND_CLAIM)
40
+ self.assertEqual(c.claim_id, "C-01")
41
+ self.assertEqual(c.kind, KIND_CLAIM)
42
+ self.assertEqual(c.tag, TAG_VERIFIED)
43
+ self.assertEqual(c.source_hash, "abc")
44
+ self.assertEqual(c.severity, "LOW")
45
+ self.assertEqual(c.status, STATUS_LIVE)
46
+
47
+ def test_normalize_claim_kind_inference(self):
48
+ raw = {
49
+ "inference_id": "I-01",
50
+ "tag": "INFERRED",
51
+ "statement": "s",
52
+ "parent_claims": ["C-01"],
53
+ "deductive_logic": "because",
54
+ }
55
+ c = normalize_claim(raw, kind=KIND_INFERENCE)
56
+ self.assertEqual(c.claim_id, "I-01")
57
+ self.assertEqual(c.kind, KIND_INFERENCE)
58
+ self.assertEqual(c.tag, TAG_INFERRED)
59
+ self.assertEqual(c.parent_claims, ["C-01"])
60
+ self.assertEqual(c.deductive_logic, "because")
61
+
62
+ def test_normalize_claim_kind_hypothesis(self):
63
+ raw = {
64
+ "hypothesis_id": "H-01",
65
+ "tag": "HYPOTHESIS",
66
+ "statement": "s",
67
+ "falsification": "measurable test",
68
+ }
69
+ c = normalize_claim(raw, kind=KIND_HYPOTHESIS)
70
+ self.assertEqual(c.claim_id, "H-01")
71
+ self.assertEqual(c.kind, KIND_HYPOTHESIS)
72
+ self.assertEqual(c.tag, TAG_HYPOTHESIS)
73
+ self.assertEqual(c.falsification, "measurable test")
74
+
75
+ def test_normalize_claim_kind_negative_knowledge(self):
76
+ raw = {"query": "q", "finding": "nothing found"}
77
+ c = normalize_claim(raw, kind=KIND_NEGATIVE_KNOWLEDGE)
78
+ self.assertEqual(c.kind, KIND_NEGATIVE_KNOWLEDGE)
79
+ self.assertEqual(c.tag, TAG_NEGATIVE_KNOWLEDGE)
80
+ self.assertEqual(c.query, "q")
81
+ self.assertEqual(c.finding, "nothing found")
82
+
83
+ def test_parent_claims_accepts_string(self):
84
+ c = normalize_claim({"inference_id": "I", "parent_claims": "C-01"}, KIND_INFERENCE)
85
+ self.assertEqual(c.parent_claims, ["C-01"])
86
+
87
+ def test_missing_tag_defaults_by_kind(self):
88
+ c = normalize_claim({"claim_id": "C"}, KIND_CLAIM)
89
+ self.assertEqual(c.tag, TAG_VERIFIED)
90
+ c = normalize_claim({"inference_id": "I"}, KIND_INFERENCE)
91
+ self.assertEqual(c.tag, TAG_INFERRED)
92
+
93
+
94
+ class TestClaimsFromDossier(unittest.TestCase):
95
+ def test_extracts_all_four_shapes(self):
96
+ dossier = {
97
+ "affirmative_claims": [{"claim_id": "A1", "statement": "s"}],
98
+ "falsification_claims": [{"claim_id": "B1", "statement": "s", "severity": "HIGH"}],
99
+ "inferred_implications": [{"inference_id": "I1", "parent_claims": ["A1"]}],
100
+ "hypotheses": [{"hypothesis_id": "H1", "falsification": "t"}],
101
+ "negative_knowledge": [{"query": "q", "finding": "f"}],
102
+ }
103
+ claims = claims_from_dossier(dossier)
104
+ self.assertEqual(len(claims), 5)
105
+ kinds = sorted(c.kind for c in claims)
106
+ self.assertEqual(
107
+ kinds,
108
+ sorted(
109
+ [
110
+ KIND_CLAIM,
111
+ KIND_CLAIM,
112
+ KIND_INFERENCE,
113
+ KIND_HYPOTHESIS,
114
+ KIND_NEGATIVE_KNOWLEDGE,
115
+ ]
116
+ ),
117
+ )
118
+
119
+ def test_handles_missing_keys(self):
120
+ self.assertEqual(claims_from_dossier({}), [])
121
+
122
+ def test_handles_non_list_rows(self):
123
+ self.assertEqual(claims_from_dossier({"affirmative_claims": "not-a-list"}), [])
124
+
125
+
126
+ class TestWitnessCheck(unittest.TestCase):
127
+ def test_missing_hash_or_quote_rejected(self):
128
+ from runner.claim_witness import ClaimWitness
129
+
130
+ class FakeHasher:
131
+ def verify_quote(self, h, q):
132
+ return True, 1.0, "ok"
133
+
134
+ c = ClaimWitness(claim_id="C", kind=KIND_CLAIM, tag=TAG_VERIFIED, statement="s")
135
+ passed, conf, msg = witness_check(c, FakeHasher())
136
+ self.assertFalse(passed)
137
+ self.assertEqual(conf, 0.0)
138
+ self.assertIn("Missing", msg)
139
+
140
+ def test_delegates_to_hasher(self):
141
+ from runner.claim_witness import ClaimWitness
142
+
143
+ calls = []
144
+
145
+ class FakeHasher:
146
+ def verify_quote(self, h, q):
147
+ calls.append((h, q))
148
+ return True, 0.99, "ok"
149
+
150
+ c = ClaimWitness(
151
+ claim_id="C",
152
+ kind=KIND_CLAIM,
153
+ tag=TAG_VERIFIED,
154
+ statement="s",
155
+ source_hash="h",
156
+ verbatim_quote="q",
157
+ )
158
+ passed, conf, msg = witness_check(c, FakeHasher())
159
+ self.assertTrue(passed)
160
+ self.assertEqual(conf, 0.99)
161
+ self.assertEqual(calls, [("h", "q")])
162
+
163
+ def test_real_hasher_round_trip(self):
164
+ import tempfile
165
+ from skills.research_cache.hasher import SourceHasher
166
+
167
+ with tempfile.TemporaryDirectory() as tmp:
168
+ hasher = SourceHasher(Path(tmp))
169
+ content = "Poseidon round constraints in 184ms with 45 GB/s."
170
+ digest = hasher.store_source(url="https://x", content=content, title="T")
171
+ c = normalize_claim(
172
+ {
173
+ "claim_id": "C",
174
+ "tag": "VERIFIED",
175
+ "statement": "s",
176
+ "source_hash": digest,
177
+ "verbatim_quote": "Poseidon round constraints in 184ms",
178
+ }
179
+ )
180
+ passed, conf, _msg = witness_check(c, hasher)
181
+ self.assertTrue(passed)
182
+ self.assertGreaterEqual(conf, 0.95)
183
+
184
+
185
+ if __name__ == "__main__":
186
+ unittest.main()