@heretek-ai/epistemic-swarm 0.6.0 → 0.7.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (81) hide show
  1. package/.claude-plugin/marketplace.json +20 -0
  2. package/.claude-plugin/plugin.json +4 -2
  3. package/MARKETPLACE.md +28 -0
  4. package/bin/cli.js +1 -0
  5. package/extensions/pi/index.js +7 -1
  6. package/install.sh +1 -1
  7. package/package.json +1 -1
  8. package/plugins/darkharvest/.claude-plugin/plugin.json +15 -0
  9. package/plugins/darkharvest/agents/harvest-proponent.md +22 -0
  10. package/plugins/darkharvest/agents/harvest-redteam.md +20 -0
  11. package/plugins/darkharvest/evals/teardown-verdict/graders/license-line.md +6 -0
  12. package/plugins/darkharvest/evals/teardown-verdict/graders/skill-fired.md +5 -0
  13. package/plugins/darkharvest/evals/teardown-verdict/prompt.md +6 -0
  14. package/plugins/darkharvest/skills/darkharvest/SKILL.md +70 -0
  15. package/plugins/darkharvest/skills/darkharvest/scripts/harvest.py +289 -0
  16. package/plugins/factory/.claude-plugin/plugin.json +15 -0
  17. package/plugins/factory/agents/factory-manager.md +22 -0
  18. package/plugins/factory/agents/programmer.md +16 -0
  19. package/plugins/factory/agents/qa-adversarial.md +17 -0
  20. package/plugins/factory/agents/qa-functional.md +17 -0
  21. package/plugins/factory/evals/gate-halt/graders/gates-first.md +6 -0
  22. package/plugins/factory/evals/gate-halt/graders/skill-fired.md +5 -0
  23. package/plugins/factory/evals/gate-halt/prompt.md +6 -0
  24. package/plugins/factory/skills/factory/SKILL.md +51 -0
  25. package/plugins/factory/skills/factory/scripts/factory.py +212 -0
  26. package/plugins/opencode/index.js +41 -1
  27. package/plugins/opencode/tui.js +66 -38
  28. package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
  29. package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
  30. package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
  31. package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
  32. package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
  33. package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
  34. package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
  35. package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
  36. package/runner/__pycache__/path_safety.cpython-311.pyc +0 -0
  37. package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
  38. package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
  39. package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
  40. package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
  41. package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
  42. package/runner/claim_store.py +15 -4
  43. package/runner/living_dossiers.py +74 -56
  44. package/runner/mcp_server.py +37 -2
  45. package/runner/pcrb.py +53 -29
  46. package/runner/refinement.py +70 -27
  47. package/runner/research_swarm.py +336 -127
  48. package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
  49. package/runner/tests/__pycache__/test_backends.cpython-311.pyc +0 -0
  50. package/runner/tests/__pycache__/test_bet1_spike.cpython-311.pyc +0 -0
  51. package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
  52. package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
  53. package/runner/tests/__pycache__/test_claude_plugin.cpython-311.pyc +0 -0
  54. package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
  55. package/runner/tests/__pycache__/test_factory.cpython-311.pyc +0 -0
  56. package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
  57. package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
  58. package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
  59. package/runner/tests/__pycache__/test_opencode_ux.cpython-311.pyc +0 -0
  60. package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
  61. package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
  62. package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
  63. package/runner/tests/__pycache__/test_sweep_regressions.cpython-311.pyc +0 -0
  64. package/runner/tests/__pycache__/test_webcache.cpython-311.pyc +0 -0
  65. package/runner/tests/test_backends.py +280 -0
  66. package/runner/tests/test_claude_plugin.py +170 -0
  67. package/runner/tests/test_sweep_regressions.py +71 -38
  68. package/scripts/__pycache__/bet1_advisory_spike.cpython-311.pyc +0 -0
  69. package/scripts/__pycache__/build_adapters.cpython-311.pyc +0 -0
  70. package/scripts/__pycache__/divergence_experiment.cpython-311.pyc +0 -0
  71. package/scripts/build_adapters.py +87 -9
  72. package/skills/darkharvest/scripts/harvest.py +76 -39
  73. package/skills/epistemic_search/scripts/__pycache__/search.cpython-311.pyc +0 -0
  74. package/skills/epistemic_search/scripts/search.py +29 -19
  75. package/skills/epistemic_search/scripts/webcache.py +29 -17
  76. package/skills/factory/scripts/factory.py +1 -1
  77. package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
  78. package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
  79. package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
  80. package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
  81. package/skills/swarm_config/configure.py +70 -37
@@ -0,0 +1,280 @@
1
+ #!/usr/bin/env python3
2
+ """Tests for host-native agent backends (parity spec section 8).
3
+
4
+ Claude Code spawns `claude -p`; OpenCode spawns `opencode run`. Default
5
+ follows the host; explicit flags/env/config always win.
6
+ """
7
+
8
+ import json
9
+ import os
10
+ import subprocess
11
+ import sys
12
+ import unittest
13
+ from pathlib import Path
14
+ from unittest import mock
15
+
16
+ PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
17
+ if str(PROJECT_ROOT) not in sys.path:
18
+ sys.path.insert(0, str(PROJECT_ROOT))
19
+
20
+ from runner.research_swarm import ( # noqa: E402
21
+ _backend_family,
22
+ _build_opencode_cmd,
23
+ _default_backend_cmd,
24
+ _extract_opencode_text,
25
+ SwarmRunner,
26
+ )
27
+
28
+
29
+ class TestBackendDefaults(unittest.TestCase):
30
+ def test_explicit_config_host_wins(self):
31
+ self.assertEqual(_default_backend_cmd("opencode"), ["opencode", "run"])
32
+ self.assertEqual(_default_backend_cmd("claude"), ["claude", "-p"])
33
+
34
+ def test_env_host_hint(self):
35
+ with mock.patch.dict(os.environ, {"IUMBTEMS_HOST": "opencode"}):
36
+ self.assertEqual(_default_backend_cmd(None), ["opencode", "run"])
37
+ with mock.patch.dict(os.environ, {"IUMBTEMS_HOST": "claude"}):
38
+ self.assertEqual(_default_backend_cmd(None), ["claude", "-p"])
39
+
40
+ def test_legacy_default_when_nothing_selects_opencode(self):
41
+ with mock.patch.dict(os.environ, {}, clear=True):
42
+ with mock.patch("shutil.which", return_value=None):
43
+ self.assertEqual(_default_backend_cmd(None), ["claude", "-p"])
44
+
45
+ def test_probe_falls_back_to_opencode_when_claude_absent(self):
46
+ with mock.patch.dict(os.environ, {}, clear=True):
47
+
48
+ def which(name):
49
+ return "/usr/bin/opencode" if name == "opencode" else None
50
+
51
+ with mock.patch("shutil.which", side_effect=which):
52
+ self.assertEqual(_default_backend_cmd(None), ["opencode", "run"])
53
+
54
+
55
+ class TestOpencodeArgv(unittest.TestCase):
56
+ def test_shape_has_no_claude_flags(self):
57
+ cmd = _build_opencode_cmd(
58
+ ["opencode", "run"], "do research", "anthropic/claude-x", "alpha-thesis"
59
+ )
60
+ self.assertEqual(cmd[0:2], ["opencode", "run"])
61
+ self.assertIn("do research", cmd)
62
+ self.assertIn("--agent", cmd)
63
+ self.assertIn("alpha-thesis", cmd)
64
+ self.assertIn("-m", cmd)
65
+ self.assertIn("anthropic/claude-x", cmd)
66
+ self.assertIn("--format", cmd)
67
+ self.assertIn("json", cmd)
68
+ self.assertNotIn("--tools", cmd)
69
+ self.assertNotIn("--system-prompt", cmd)
70
+ self.assertFalse(any(a.startswith("--model") for a in cmd))
71
+
72
+ def test_minimal_shape(self):
73
+ self.assertEqual(
74
+ _build_opencode_cmd(["opencode", "run"], "p", None, None),
75
+ ["opencode", "run", "p", "--format", "json"],
76
+ )
77
+
78
+ def test_family_detection(self):
79
+ self.assertEqual(_backend_family(["opencode", "run"]), "opencode")
80
+ self.assertEqual(_backend_family(["claude", "-p"]), "claude")
81
+ self.assertEqual(_backend_family(["/usr/bin/opencode", "run"]), "opencode")
82
+
83
+
84
+ class TestRunnerBackendWiring(unittest.TestCase):
85
+ def _runner(self, **kw):
86
+ import tempfile
87
+
88
+ tmp = tempfile.mkdtemp()
89
+ return SwarmRunner(
90
+ base_dir=Path(tmp), mock_mode=True, mode=kw.pop("mode", "research"), **kw
91
+ )
92
+
93
+ def test_claude_shape_preserved_by_default(self):
94
+ with mock.patch.dict(os.environ, {}, clear=True):
95
+ with mock.patch("shutil.which", return_value=None):
96
+ r = self._runner()
97
+ cmd = r.build_agent_cmd("obj")
98
+ self.assertEqual(cmd[:3], ["claude", "-p", "obj"])
99
+ self.assertIn("--tools", cmd)
100
+
101
+ def test_opencode_shape_via_host_env(self):
102
+ with mock.patch.dict(os.environ, {"IUMBTEMS_HOST": "opencode"}):
103
+ r = self._runner(mode="scout")
104
+ cmd = r.build_agent_cmd("obj")
105
+ self.assertEqual(cmd[:2], ["opencode", "run"])
106
+ self.assertIn("oss-scout", cmd)
107
+
108
+ def test_research_beta_maps_to_beta_redteam(self):
109
+ with mock.patch.dict(os.environ, {"IUMBTEMS_HOST": "opencode"}):
110
+ r = self._runner(mode="research")
111
+ self.assertIn("beta-redteam", r.build_agent_cmd("obj", role="beta"))
112
+ self.assertIn("alpha-thesis", r.build_agent_cmd("obj", role="alpha"))
113
+
114
+ def test_explicit_override_beats_host_env(self):
115
+ with mock.patch.dict(os.environ, {"IUMBTEMS_HOST": "opencode"}):
116
+ r = self._runner(agent_overrides={"alpha": {"backend": ["claude", "-p"]}})
117
+ cmd = r.build_agent_cmd("obj", role="alpha")
118
+ self.assertEqual(cmd[:2], ["claude", "-p"])
119
+
120
+
121
+ class TestOpencodeOutputParsing(unittest.TestCase):
122
+ def test_extracts_text_events(self):
123
+ raw = "\n".join(
124
+ [
125
+ '{"type":"session","id":"s1"}',
126
+ '{"type":"message","text":"first finding"}',
127
+ '{"type":"message","text":"second finding"}',
128
+ '{"type":"tool_call","tool":"read"}',
129
+ ]
130
+ )
131
+ self.assertEqual(_extract_opencode_text(raw), "first finding\nsecond finding")
132
+
133
+ def test_falls_back_to_raw(self):
134
+ self.assertEqual(_extract_opencode_text("plain text out"), "plain text out")
135
+ self.assertEqual(_extract_opencode_text(""), "")
136
+
137
+ def test_ignores_malformed_lines(self):
138
+ raw = '{not json\n{"type":"result","result":"verdict: clean-room"}'
139
+ self.assertEqual(_extract_opencode_text(raw), "verdict: clean-room")
140
+
141
+
142
+ class TestDossierStdoutFallback(unittest.TestCase):
143
+ def _runner(self):
144
+ import tempfile
145
+ from runner.research_swarm import SwarmRunner
146
+
147
+ tmp = tempfile.mkdtemp()
148
+ r = SwarmRunner(base_dir=Path(tmp), mock_mode=False, mode="research")
149
+ r.state_machine.init_session("fallback probe")
150
+ r.state_machine.set_scopes(
151
+ [
152
+ {
153
+ "scope_id": "scope_01_probe",
154
+ "title": "Probe",
155
+ "objective": "probe objective",
156
+ "dependencies": [],
157
+ "affirmative_targets": [],
158
+ "adversarial_targets": [],
159
+ }
160
+ ]
161
+ )
162
+ return r
163
+
164
+ def test_recovers_dossier_from_stdout(self):
165
+ from runner.research_swarm import SwarmRunner
166
+
167
+ r = self._runner()
168
+ payload = json.dumps(
169
+ {
170
+ "agent": "Agent Alpha (Thesis)",
171
+ "scope_id": "scope_01_probe",
172
+ "affirmative_claims": [],
173
+ }
174
+ )
175
+ transcript = "Some chatter\n```json\n" + payload + "\n```\nDone."
176
+ with mock.patch.object(
177
+ SwarmRunner, "run_claude_process", return_value=transcript
178
+ ):
179
+ scope = {
180
+ "scope_id": "scope_01_probe",
181
+ "title": "Probe",
182
+ "objective": "probe objective",
183
+ "dependencies": [],
184
+ "affirmative_targets": [],
185
+ "adversarial_targets": [],
186
+ }
187
+ r.run_agent_alpha(scope)
188
+ dossier_path = (
189
+ r.state_machine.get_scope_dir("scope_01_probe") / "alpha_dossier.json"
190
+ )
191
+ self.assertTrue(dossier_path.is_file())
192
+ saved = json.loads(dossier_path.read_text())
193
+ self.assertTrue(saved.get("recovered_from_stdout"))
194
+
195
+ def test_raises_when_no_file_and_no_json(self):
196
+ from runner.research_swarm import SwarmRunner
197
+
198
+ r = self._runner()
199
+ with mock.patch.object(
200
+ SwarmRunner, "run_claude_process", return_value="just some prose, no json"
201
+ ):
202
+ scope = {
203
+ "scope_id": "scope_01_probe",
204
+ "title": "Probe",
205
+ "objective": "probe objective",
206
+ "dependencies": [],
207
+ "affirmative_targets": [],
208
+ "adversarial_targets": [],
209
+ }
210
+ with self.assertRaises(FileNotFoundError):
211
+ r.run_agent_beta(scope)
212
+
213
+
214
+ class TestProjectDirResolution(unittest.TestCase):
215
+ def test_explicit_base_dir_wins(self):
216
+ from runner.mcp_server import _resolve_base_dir
217
+
218
+ with mock.patch.dict(os.environ, {"IUMBTEMS_PROJECT_DIR": "/tmp"}):
219
+ p = _resolve_base_dir({"base_dir": "/tmp/custom"})
220
+ self.assertEqual(p, Path("/tmp/custom"))
221
+
222
+ def test_project_dir_preferred_over_cwd(self):
223
+ import tempfile
224
+ from runner.mcp_server import _resolve_base_dir
225
+
226
+ project = tempfile.mkdtemp()
227
+ with mock.patch.dict(os.environ, {"IUMBTEMS_PROJECT_DIR": project}):
228
+ p = _resolve_base_dir({})
229
+ self.assertEqual(
230
+ p, Path(os.path.realpath(os.path.join(project, ".research")))
231
+ )
232
+
233
+ def test_missing_project_dir_falls_back_to_cwd(self):
234
+ from runner.mcp_server import _resolve_base_dir
235
+
236
+ with mock.patch.dict(os.environ, {"IUMBTEMS_PROJECT_DIR": "/no/such/dir"}):
237
+ p = _resolve_base_dir({})
238
+ self.assertEqual(p, Path(os.path.realpath(".research")))
239
+
240
+
241
+ def run_node(code):
242
+ return subprocess.run(
243
+ ["node", "--input-type=module", "-e", code],
244
+ capture_output=True,
245
+ text=True,
246
+ cwd=str(PROJECT_ROOT),
247
+ )
248
+
249
+
250
+ def last_json_object(stdout):
251
+ lines = [l.strip() for l in stdout.strip().split("\n") if l.strip().startswith("{")]
252
+ return json.loads(lines[-1])
253
+
254
+
255
+ class TestOpencodePluginBackend(unittest.TestCase):
256
+ def test_backend_input_mirrored_on_swarm_tools(self):
257
+ res = run_node(
258
+ """
259
+ import plugin from "./plugins/opencode/index.js";
260
+ const toolMap = {};
261
+ const host = {
262
+ options: {},
263
+ command: {list: async () => ({data: []}), transform: async () => ({dispose: () => {}}), reload: async () => {}},
264
+ tool: {transform: async (fn) => { fn({add: (t) => { toolMap[t.name] = t; }}); return {dispose: () => {}}; }, reload: async () => {}},
265
+ session: {prompt: async () => ({})}
266
+ };
267
+ await plugin.setup(host);
268
+ const names = ["iumbtems_swarm_research","iumbtems_code_audit","iumbtems_oss_scout","iumbtems_brainstorm","iumbtems_darkharvest"];
269
+ console.log(JSON.stringify(Object.fromEntries(names.map(n => [n, toolMap[n]?.input?.properties?.backend?.enum || null]))));
270
+ """
271
+ )
272
+ self.assertEqual(res.returncode, 0, res.stderr)
273
+ data = last_json_object(res.stdout)
274
+ self.assertEqual(len(data), 5)
275
+ for name, enum in data.items():
276
+ self.assertEqual(enum, ["auto", "claude", "opencode"], name)
277
+
278
+
279
+ if __name__ == "__main__":
280
+ unittest.main()
@@ -0,0 +1,170 @@
1
+ #!/usr/bin/env python3
2
+ """Tests for Claude Code plugin surfaces: manifests, agents, modular sync, evals.
3
+
4
+ Per https://code.claude.com/docs/en/plugins/create (manifest + layout),
5
+ /components (skills/agents/hooks/MCP), and /plugin-evals (suite layout).
6
+ Behavioral eval runs are billable and stay in CI (plugin-evals.yml); here we
7
+ assert suite structure only.
8
+ """
9
+
10
+ import json
11
+ import unittest
12
+ from pathlib import Path
13
+
14
+ PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
15
+ if str(PROJECT_ROOT) not in __import__("sys").path:
16
+ __import__("sys").path.insert(0, str(PROJECT_ROOT))
17
+
18
+ MODULAR = {
19
+ "socratic-grilling": "grilling",
20
+ "research-cache": "research_cache",
21
+ "darkharvest": "darkharvest",
22
+ "factory": "factory",
23
+ }
24
+
25
+
26
+ def parse_frontmatter(path):
27
+ text = Path(path).read_text(encoding="utf-8")
28
+ if not text.startswith("---"):
29
+ return {}
30
+ end = text.find("---", 3)
31
+ if end < 0:
32
+ return {}
33
+ data = {}
34
+ for line in text[3:end].strip().splitlines():
35
+ if ":" in line:
36
+ k, v = line.split(":", 1)
37
+ data[k.strip()] = v.strip()
38
+ return data
39
+
40
+
41
+ class TestClaudeManifests(unittest.TestCase):
42
+ def test_root_manifest_lists_all_nine_skills(self):
43
+ manifest = json.loads(
44
+ (PROJECT_ROOT / ".claude-plugin" / "plugin.json").read_text()
45
+ )
46
+ self.assertEqual(manifest["name"], "epistemic-swarm")
47
+ for skill in (
48
+ "grilling",
49
+ "research_cache",
50
+ "epistemic_search",
51
+ "swarm_config",
52
+ "code_audit",
53
+ "oss_scout",
54
+ "brainstorming",
55
+ "darkharvest",
56
+ "factory",
57
+ ):
58
+ self.assertIn(f"./skills/{skill}", manifest["skills"])
59
+
60
+ def test_marketplace_entries_resolve(self):
61
+ marketplace = json.loads(
62
+ (PROJECT_ROOT / ".claude-plugin" / "marketplace.json").read_text()
63
+ )
64
+ names = [p["name"] for p in marketplace["plugins"]]
65
+ for name in (
66
+ "epistemic-swarm",
67
+ "socratic-grilling",
68
+ "research-cache",
69
+ "darkharvest",
70
+ "factory",
71
+ ):
72
+ self.assertIn(name, names)
73
+ for plugin in marketplace["plugins"]:
74
+ if plugin["name"] == "epistemic-swarm":
75
+ continue
76
+ plugin_json = (
77
+ PROJECT_ROOT
78
+ / plugin["source"].lstrip("./")
79
+ / ".claude-plugin"
80
+ / "plugin.json"
81
+ )
82
+ self.assertTrue(plugin_json.is_file(), f"missing {plugin_json}")
83
+ data = json.loads(plugin_json.read_text())
84
+ self.assertEqual(data["name"], plugin["name"])
85
+
86
+ def test_modular_skills_in_sync(self):
87
+ for mod, skill in MODULAR.items():
88
+ src = PROJECT_ROOT / "skills" / skill
89
+ dst = PROJECT_ROOT / "plugins" / mod / "skills" / skill
90
+ self.assertTrue((dst / "SKILL.md").is_file(), f"missing {dst}")
91
+ src_files = {
92
+ p.relative_to(src)
93
+ for p in src.rglob("*")
94
+ if p.is_file() and "__pycache__" not in p.parts
95
+ }
96
+ dst_files = {
97
+ p.relative_to(dst)
98
+ for p in dst.rglob("*")
99
+ if p.is_file() and "__pycache__" not in p.parts
100
+ }
101
+ self.assertEqual(src_files, dst_files, f"drift in {mod}")
102
+ for rel in src_files:
103
+ self.assertEqual(
104
+ (src / rel).read_bytes(),
105
+ (dst / rel).read_bytes(),
106
+ f"content drift: {mod}/{rel}",
107
+ )
108
+
109
+
110
+ class TestClaudeAgents(unittest.TestCase):
111
+ AGENTS = {
112
+ "agents/alpha-thesis.md": "alpha-thesis",
113
+ "agents/beta-antithesis.md": "beta-antithesis",
114
+ "agents/epistemic-auditor.md": "epistemic-auditor",
115
+ "plugins/darkharvest/agents/harvest-proponent.md": "harvest-proponent",
116
+ "plugins/darkharvest/agents/harvest-redteam.md": "harvest-redteam",
117
+ "plugins/factory/agents/factory-manager.md": "factory-manager",
118
+ "plugins/factory/agents/programmer.md": "programmer",
119
+ "plugins/factory/agents/qa-functional.md": "qa-functional",
120
+ "plugins/factory/agents/qa-adversarial.md": "qa-adversarial",
121
+ }
122
+
123
+ def test_agent_frontmatter(self):
124
+ for rel, expected_name in self.AGENTS.items():
125
+ fm = parse_frontmatter(PROJECT_ROOT / rel)
126
+ self.assertEqual(fm.get("name"), expected_name, rel)
127
+ self.assertTrue(fm.get("description"), f"no description: {rel}")
128
+ self.assertIn("model", fm, f"no model: {rel}")
129
+ for banned in ("permissionMode", "hooks", "mcpServers", "initialPrompt"):
130
+ self.assertNotIn(banned, fm, f"banned field in {rel}")
131
+
132
+
133
+ class TestClaudeEvals(unittest.TestCase):
134
+ CASES = {
135
+ "evals/grill-fires": "grilling",
136
+ "evals/darkharvest-fires": "darkharvest",
137
+ "evals/factory-gate": "factory",
138
+ "plugins/darkharvest/evals/teardown-verdict": "darkharvest",
139
+ "plugins/factory/evals/gate-halt": "factory",
140
+ }
141
+
142
+ def test_eval_suite_structure(self):
143
+ for case, skill in self.CASES.items():
144
+ prompt = PROJECT_ROOT / case / "prompt.md"
145
+ self.assertTrue(prompt.is_file(), f"missing {prompt}")
146
+ body = prompt.read_text(encoding="utf-8")
147
+ self.assertIn("allowed_tools", body)
148
+ graders = sorted((PROJECT_ROOT / case / "graders").glob("*.md"))
149
+ self.assertTrue(graders, f"no graders in {case}")
150
+ kinds = set()
151
+ for g in graders:
152
+ fm = parse_frontmatter(g)
153
+ self.assertIn(
154
+ fm.get("type"),
155
+ (
156
+ "tool_used",
157
+ "llm",
158
+ "regex",
159
+ "tool_order",
160
+ "file_exists",
161
+ "baseline",
162
+ ),
163
+ f"bad type in {g}",
164
+ )
165
+ kinds.add(fm.get("type"))
166
+ self.assertIn("tool_used", kinds, f"no Skill-fired grader in {case}")
167
+
168
+
169
+ if __name__ == "__main__":
170
+ unittest.main()
@@ -64,11 +64,9 @@ class TestBuildAdaptersImportable(unittest.TestCase):
64
64
  # Python 3.14 (PEP 649) defers annotation evaluation, so the NameError
65
65
  # only surfaces when something actually resolves the annotations. We
66
66
  # force that resolution with get_type_hints() so the test reproduces
67
- # the CI failure on every interpreter.
68
- try:
69
- module = _load_module("scripts/build_adapters.py", "iumbtems_build_adapters")
70
- except NameError as e:
71
- self.fail(f"scripts/build_adapters.py fails to import on eager-annotation Python: {e}")
67
+ # the CI failure on every interpreter. An import-time NameError
68
+ # propagates uncaught (red test) — no wrapper, per S8714.
69
+ module = _load_module("scripts/build_adapters.py", "iumbtems_build_adapters")
72
70
 
73
71
  unresolved = []
74
72
  for attr_name, attr in list(vars(module).items()):
@@ -86,19 +84,26 @@ class TestBuildAdaptersImportable(unittest.TestCase):
86
84
  # an annotation defect.
87
85
  continue
88
86
  self.assertEqual(
89
- unresolved, [],
87
+ unresolved,
88
+ [],
90
89
  "annotations reference undefined names (this is the Python 3.11 "
91
90
  f"import-time NameError): {unresolved}",
92
91
  )
93
92
 
94
93
  def test_check_mode_exits_zero(self):
95
94
  res = subprocess.run(
96
- [sys.executable, str(PROJECT_ROOT / "scripts" / "build_adapters.py"), "--check"],
95
+ [
96
+ sys.executable,
97
+ str(PROJECT_ROOT / "scripts" / "build_adapters.py"),
98
+ "--check",
99
+ ],
97
100
  capture_output=True,
98
101
  text=True,
99
102
  cwd=str(PROJECT_ROOT),
100
103
  )
101
- self.assertEqual(res.returncode, 0, f"--check failed:\n{res.stdout}\n{res.stderr}")
104
+ self.assertEqual(
105
+ res.returncode, 0, f"--check failed:\n{res.stdout}\n{res.stderr}"
106
+ )
102
107
 
103
108
 
104
109
  class TestSearchParserRegression(unittest.TestCase):
@@ -124,7 +129,7 @@ class TestSearchParserRegression(unittest.TestCase):
124
129
  and zip(links, snippets) produced nothing.
125
130
  """
126
131
  html = (
127
- '<table><tr>'
132
+ "<table><tr>"
128
133
  '<td><a class="result-link" href="https://a.example/x">Title</a></td>'
129
134
  '<td class="result-snippet">See '
130
135
  '<a href="/l/?uddg=https%3A%2F%2Fy.example" class="result-link">this</a> page'
@@ -167,17 +172,17 @@ class TestSearchParserRegression(unittest.TestCase):
167
172
  A bare `&` from `&amp;` (or from a URL query string) makes the
168
173
  <search_results> document malformed.
169
174
  """
170
- xml_text = self.search.format_xml([
171
- {
172
- "title": "Q&A",
173
- "url": "https://x.example/?a=1&b=2",
174
- "snippet": "A & B and C++ <tips>",
175
- }
176
- ])
177
- try:
178
- xml.dom.minidom.parseString(xml_text)
179
- except Exception as e: # pragma: no cover - failure path
180
- self.fail(f"format_xml produced malformed XML: {e}\n{xml_text}")
175
+ xml_text = self.search.format_xml(
176
+ [
177
+ {
178
+ "title": "Q&A",
179
+ "url": "https://x.example/?a=1&b=2",
180
+ "snippet": "A & B and C++ <tips>",
181
+ }
182
+ ]
183
+ )
184
+ # Malformed output raises here (red test) — no wrapper, per S8714.
185
+ xml.dom.minidom.parseString(xml_text)
181
186
  self.assertIn("&amp;", xml_text)
182
187
  self.assertIn("&lt;tips&gt;", xml_text)
183
188
 
@@ -192,11 +197,15 @@ class TestSearchParserRegression(unittest.TestCase):
192
197
  results = self._parse(html)
193
198
  self.assertEqual(results[0]["title"], "Q&A")
194
199
  self.assertEqual(results[0]["snippet"], "A & B")
195
- xml_text = self.search.format_xml([{
196
- "title": results[0]["title"],
197
- "url": "https://a.example/x",
198
- "snippet": results[0]["snippet"],
199
- }])
200
+ xml_text = self.search.format_xml(
201
+ [
202
+ {
203
+ "title": results[0]["title"],
204
+ "url": "https://a.example/x",
205
+ "snippet": results[0]["snippet"],
206
+ }
207
+ ]
208
+ )
200
209
  xml.dom.minidom.parseString(xml_text) # raises if malformed
201
210
  self.assertIn("<title>Q&amp;A</title>", xml_text)
202
211
  self.assertIn("<snippet>A &amp; B</snippet>", xml_text)
@@ -222,7 +231,9 @@ class TestDivergenceAggregation(unittest.TestCase):
222
231
  def test_single_scope_is_unchanged(self):
223
232
  with tempfile.TemporaryDirectory() as tmp:
224
233
  base = Path(tmp)
225
- self._write_audits(base, {"scope_a": {"divergence_score": 0.5, "verified_passed": True}})
234
+ self._write_audits(
235
+ base, {"scope_a": {"divergence_score": 0.5, "verified_passed": True}}
236
+ )
226
237
  score, verified = self.div._extract_audit_divergence(base)
227
238
  self.assertAlmostEqual(score, 0.5)
228
239
  self.assertTrue(verified)
@@ -231,11 +242,14 @@ class TestDivergenceAggregation(unittest.TestCase):
231
242
  """The rule is the mean — independent of scope-name sort order."""
232
243
  with tempfile.TemporaryDirectory() as tmp:
233
244
  base = Path(tmp)
234
- self._write_audits(base, {
235
- "aaa_first": {"divergence_score": 0.0, "verified_passed": True},
236
- "mmm_mid": {"divergence_score": 0.5, "verified_passed": True},
237
- "zzz_last": {"divergence_score": 1.0, "verified_passed": False},
238
- })
245
+ self._write_audits(
246
+ base,
247
+ {
248
+ "aaa_first": {"divergence_score": 0.0, "verified_passed": True},
249
+ "mmm_mid": {"divergence_score": 0.5, "verified_passed": True},
250
+ "zzz_last": {"divergence_score": 1.0, "verified_passed": False},
251
+ },
252
+ )
239
253
  score, verified = self.div._extract_audit_divergence(base)
240
254
  self.assertAlmostEqual(score, 0.5)
241
255
  # Not the first scope's 0.0, not the last scope's 1.0.
@@ -247,10 +261,13 @@ class TestDivergenceAggregation(unittest.TestCase):
247
261
  def test_unscored_audits_are_skipped(self):
248
262
  with tempfile.TemporaryDirectory() as tmp:
249
263
  base = Path(tmp)
250
- self._write_audits(base, {
251
- "aaa": {"verified_passed": True}, # no score
252
- "bbb": {"divergence_score": 0.25, "verified_passed": True},
253
- })
264
+ self._write_audits(
265
+ base,
266
+ {
267
+ "aaa": {"verified_passed": True}, # no score
268
+ "bbb": {"divergence_score": 0.25, "verified_passed": True},
269
+ },
270
+ )
254
271
  score, verified = self.div._extract_audit_divergence(base)
255
272
  self.assertAlmostEqual(score, 0.25)
256
273
  self.assertTrue(verified)
@@ -294,7 +311,9 @@ class TestBackendValidation(unittest.TestCase):
294
311
 
295
312
  def test_bad_backend_becomes_runtime_error_not_propagated_valueerror(self):
296
313
  """A bad backend must surface as a run failure, not abort the swarm."""
297
- runner = self.swarm.SwarmRunner(mock_mode=False, base_dir=Path(tempfile.mkdtemp()))
314
+ runner = self.swarm.SwarmRunner(
315
+ mock_mode=False, base_dir=Path(tempfile.mkdtemp())
316
+ )
298
317
  original = runner.build_agent_cmd
299
318
  runner.build_agent_cmd = lambda *a, **k: ["definitely-not-a-real-binary-xyz"]
300
319
  try:
@@ -348,17 +367,20 @@ class TestKeyFileFailsLoudly(unittest.TestCase):
348
367
 
349
368
  def test_pcrb_resolve_key_rejects_missing_file(self):
350
369
  from runner.pcrb import _resolve_key
370
+
351
371
  with self.assertRaises(ValueError):
352
372
  _resolve_key(None, Path("/definitely/not/a/real/key/file"))
353
373
 
354
374
  def test_pcrb_resolve_key_rejects_directory(self):
355
375
  from runner.pcrb import _resolve_key
376
+
356
377
  with tempfile.TemporaryDirectory() as tmp:
357
378
  with self.assertRaises(ValueError):
358
379
  _resolve_key(None, Path(tmp))
359
380
 
360
381
  def test_pcrb_resolve_key_reads_valid_file(self):
361
382
  from runner.pcrb import _resolve_key
383
+
362
384
  with tempfile.TemporaryDirectory() as tmp:
363
385
  kf = Path(tmp) / "key.txt"
364
386
  kf.write_text("secret\n", encoding="utf-8")
@@ -366,6 +388,7 @@ class TestKeyFileFailsLoudly(unittest.TestCase):
366
388
 
367
389
  def test_pcrb_resolve_key_absent_returns_none(self):
368
390
  from runner.pcrb import _resolve_key
391
+
369
392
  self.assertIsNone(_resolve_key(None, None))
370
393
 
371
394
 
@@ -384,7 +407,16 @@ class TestNoStaleV1PluginSurface(unittest.TestCase):
384
407
  "registerOpenCodeCommands",
385
408
  "experimental.session.compacting",
386
409
  )
387
- SCAN_DIRS = (".github", "scripts", "bin", "config", "runner", "extensions", "docs", "skills")
410
+ SCAN_DIRS = (
411
+ ".github",
412
+ "scripts",
413
+ "bin",
414
+ "config",
415
+ "runner",
416
+ "extensions",
417
+ "docs",
418
+ "skills",
419
+ )
388
420
  SCAN_SUFFIXES = (".yml", ".yaml", ".js", ".ts", ".py", ".sh", ".json", ".md")
389
421
 
390
422
  def test_no_v1_surface_references(self):
@@ -410,7 +442,8 @@ class TestNoStaleV1PluginSurface(unittest.TestCase):
410
442
  if pattern in text:
411
443
  offenders.append(f"{path.relative_to(PROJECT_ROOT)}: {pattern}")
412
444
  self.assertEqual(
413
- offenders, [],
445
+ offenders,
446
+ [],
414
447
  "references to the removed V1 opencode plugin surface "
415
448
  "(these broke CI when `server()` was dropped): " + "; ".join(offenders),
416
449
  )