@heretek-ai/epistemic-swarm 0.2.2 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (138) hide show
  1. package/.agents/skills/brainstorming/SKILL.md +13 -0
  2. package/.agents/skills/code_audit/SKILL.md +13 -0
  3. package/.agents/skills/epistemic_search/SKILL.md +13 -0
  4. package/.agents/skills/grilling/SKILL.md +13 -0
  5. package/.agents/skills/oss_scout/SKILL.md +13 -0
  6. package/.agents/skills/research_cache/SKILL.md +13 -0
  7. package/.agents/skills/swarm_config/SKILL.md +13 -0
  8. package/.claude-plugin/plugin.json +15 -5
  9. package/.omp/README.md +39 -0
  10. package/.omp/SYSTEM.md +12 -0
  11. package/.omp/commands/audit.md +10 -0
  12. package/.omp/commands/brainstorming.md +12 -0
  13. package/.omp/commands/grill.md +9 -0
  14. package/.omp/commands/scout.md +10 -0
  15. package/.omp/commands/swarm-config.md +10 -0
  16. package/.omp/commands/swarm.md +9 -0
  17. package/.omp/hooks/post/epistemic-audit.ts +16 -0
  18. package/.omp/hooks/pre/epistemic-redirect.ts +21 -0
  19. package/.omp/prompts/brainstorming.md +8 -0
  20. package/.omp/prompts/swarm.md +8 -0
  21. package/MARKETPLACE.md +8 -0
  22. package/README.md +53 -8
  23. package/bin/cli.js +27 -3
  24. package/config/domain_packs/biopharma.json +23 -0
  25. package/config/domain_packs/legal.json +19 -0
  26. package/config/domain_packs/quant.json +19 -0
  27. package/config/mcp-research-servers.json +7 -0
  28. package/config/mcp_launcher.py +48 -137
  29. package/config/opencode-snippet.json +58 -3
  30. package/config/searxng_mcp.py +42 -83
  31. package/extensions/pi/index.js +196 -28
  32. package/install.sh +20 -4
  33. package/package.json +40 -5
  34. package/plugins/antigravity/README.md +28 -0
  35. package/plugins/antigravity/agents/alpha-thesis.md +6 -0
  36. package/plugins/antigravity/agents/beta-antithesis.md +7 -0
  37. package/plugins/antigravity/agents/brainstormer.md +7 -0
  38. package/plugins/antigravity/agents/epistemic-auditor.md +5 -0
  39. package/plugins/antigravity/hooks.json +23 -0
  40. package/plugins/antigravity/mcp_config.json +33 -0
  41. package/plugins/antigravity/plugin.json +21 -0
  42. package/plugins/antigravity/rules/epistemic-integrity.md +6 -0
  43. package/plugins/antigravity/skills/brainstorming/SKILL.md +13 -0
  44. package/plugins/antigravity/skills/code_audit/SKILL.md +13 -0
  45. package/plugins/antigravity/skills/epistemic_search/SKILL.md +13 -0
  46. package/plugins/antigravity/skills/grilling/SKILL.md +13 -0
  47. package/plugins/antigravity/skills/oss_scout/SKILL.md +13 -0
  48. package/plugins/antigravity/skills/research_cache/SKILL.md +13 -0
  49. package/plugins/antigravity/skills/swarm_config/SKILL.md +13 -0
  50. package/plugins/codex/AGENTS.md.snippet +10 -0
  51. package/plugins/codex/README.md +37 -0
  52. package/plugins/codex/config.toml.snippet +28 -0
  53. package/plugins/codex/openai.yaml +24 -0
  54. package/plugins/codex/skills/brainstorming/SKILL.md +13 -0
  55. package/plugins/codex/skills/code_audit/SKILL.md +13 -0
  56. package/plugins/codex/skills/epistemic_search/SKILL.md +13 -0
  57. package/plugins/codex/skills/grilling/SKILL.md +13 -0
  58. package/plugins/codex/skills/oss_scout/SKILL.md +13 -0
  59. package/plugins/codex/skills/research_cache/SKILL.md +13 -0
  60. package/plugins/codex/skills/swarm_config/SKILL.md +13 -0
  61. package/plugins/gemini/GEMINI.md +15 -0
  62. package/plugins/gemini/README.md +19 -0
  63. package/plugins/gemini/commands/audit.toml +6 -0
  64. package/plugins/gemini/commands/brainstorming.toml +10 -0
  65. package/plugins/gemini/commands/grill.toml +6 -0
  66. package/plugins/gemini/commands/scout.toml +7 -0
  67. package/plugins/gemini/commands/swarm-config.toml +7 -0
  68. package/plugins/gemini/commands/swarm.toml +8 -0
  69. package/plugins/gemini/gemini-extension.json +38 -0
  70. package/plugins/gemini/hooks/hooks.json +11 -0
  71. package/plugins/gemini/skills/brainstorming/SKILL.md +13 -0
  72. package/plugins/gemini/skills/code_audit/SKILL.md +13 -0
  73. package/plugins/gemini/skills/epistemic_search/SKILL.md +13 -0
  74. package/plugins/gemini/skills/grilling/SKILL.md +13 -0
  75. package/plugins/gemini/skills/oss_scout/SKILL.md +13 -0
  76. package/plugins/gemini/skills/research_cache/SKILL.md +13 -0
  77. package/plugins/gemini/skills/swarm_config/SKILL.md +13 -0
  78. package/plugins/opencode/index.js +335 -118
  79. package/prompts/agent_brainstormer.md +97 -0
  80. package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
  81. package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
  82. package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
  83. package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
  84. package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
  85. package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
  86. package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
  87. package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
  88. package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
  89. package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
  90. package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
  91. package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
  92. package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
  93. package/runner/auctioneer.py +169 -0
  94. package/runner/auditor_engine.py +127 -12
  95. package/runner/claim_store.py +361 -0
  96. package/runner/claim_witness.py +183 -0
  97. package/runner/living_dossiers.py +355 -0
  98. package/runner/mcp_protocol.py +188 -0
  99. package/runner/mcp_server.py +567 -0
  100. package/runner/pcrb.py +212 -0
  101. package/runner/pcrb_verify.py +237 -0
  102. package/runner/refinement.py +335 -0
  103. package/runner/research_swarm.py +378 -94
  104. package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
  105. package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
  106. package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
  107. package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
  108. package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
  109. package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
  110. package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
  111. package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
  112. package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
  113. package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
  114. package/runner/tests/fixtures/auction_objective.json +35 -0
  115. package/runner/tests/fixtures/borderline_claims.json +27 -0
  116. package/runner/tests/fixtures/divergence_objectives.json +31 -0
  117. package/runner/tests/test_auction_order.py +173 -0
  118. package/runner/tests/test_claim_store.py +162 -0
  119. package/runner/tests/test_claim_witness.py +186 -0
  120. package/runner/tests/test_domain_packs.py +217 -0
  121. package/runner/tests/test_fleet_seam.py +113 -0
  122. package/runner/tests/test_living_dossiers.py +204 -0
  123. package/runner/tests/test_mcp_server.py +212 -0
  124. package/runner/tests/test_pcrb.py +240 -0
  125. package/runner/tests/test_refinement.py +255 -0
  126. package/runner/tests/test_swarm.py +173 -15
  127. package/scripts/auction_experiment.py +180 -0
  128. package/scripts/build_adapters.py +183 -0
  129. package/scripts/divergence_experiment.py +184 -0
  130. package/skills/brainstorming/SKILL.md +106 -0
  131. package/skills/brainstorming/__init__.py +1 -0
  132. package/skills/brainstorming/scripts/brainstorm.py +200 -0
  133. package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
  134. package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
  135. package/skills/swarm_config/SKILL.md +1 -1
  136. package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
  137. package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
  138. package/skills/swarm_config/configure.py +45 -11
@@ -0,0 +1,180 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ Stream F / Frontier Markets probe harness.
4
+
5
+ Probe: [HYPOTHESIS: on a fixed 4-scope objective, auction allocation matches
6
+ static-DAG wall-clock and yields >= 20% more distinct verified claims per
7
+ token]
8
+
9
+ Runs the same fixed objective twice — once with allocation="dag", once with
10
+ allocation="auction" — and emits a comparison artifact under
11
+ .research/experiments/. LIVE MODE SPENDS MONEY; use --mock to prove the
12
+ harness (the probe itself requires real models).
13
+
14
+ Token accounting is a chars/4 ESTIMATE (flagged approximation) — report both
15
+ chars and tokens in the artifact.
16
+
17
+ Usage:
18
+ python3 scripts/auction_experiment.py --mock # harness proof
19
+ python3 scripts/auction_experiment.py # live probe run
20
+ """
21
+
22
+ import argparse
23
+ import contextlib
24
+ import io
25
+ import json
26
+ import sys
27
+ import time
28
+ from datetime import datetime, timezone
29
+ from pathlib import Path
30
+ from typing import Any, Dict, Optional
31
+
32
+ PROJECT_ROOT = Path(__file__).resolve().parent.parent
33
+ if str(PROJECT_ROOT) not in sys.path:
34
+ sys.path.insert(0, str(PROJECT_ROOT))
35
+
36
+ from runner.auctioneer import estimate_tokens # noqa: E402
37
+ from runner.research_swarm import SwarmRunner # noqa: E402
38
+
39
+ FIXTURE = PROJECT_ROOT / "runner" / "tests" / "fixtures" / "auction_objective.json"
40
+ GATE_BAR_CLAIMS_PER_TOKEN = 0.20
41
+
42
+
43
+ def load_fixture(path: Path = FIXTURE) -> Dict[str, Any]:
44
+ with open(path, "r", encoding="utf-8") as f:
45
+ return json.load(f)
46
+
47
+
48
+ def run_once(
49
+ fixture: Dict[str, Any],
50
+ base_dir: Path,
51
+ allocation: str,
52
+ mock: bool,
53
+ ) -> Dict[str, Any]:
54
+ """One full swarm run under one allocation policy. Returns metrics."""
55
+ # Seed a fixed 4-scope manifest by pre-writing the scope list through a
56
+ # stub orchestrator: we pass scopes via a tiny monkeypatch so both arms
57
+ # see the IDENTICAL scope set (fair comparison).
58
+ runner = SwarmRunner(base_dir=base_dir, mock_mode=mock, allocation=allocation)
59
+
60
+ fixed_scopes = fixture["scopes"]
61
+
62
+ def fixed_orchestrate(objective, frontier_file=None):
63
+ runner.state_machine.init_session(objective)
64
+ runner.state_machine.set_scopes(fixed_scopes)
65
+ return fixed_scopes
66
+
67
+ runner.orchestrate_objective = fixed_orchestrate # type: ignore[method-assign]
68
+
69
+ buf = io.StringIO()
70
+ started = time.perf_counter()
71
+ with contextlib.redirect_stdout(buf):
72
+ runner.run_swarm(fixture["objective"])
73
+ elapsed = time.perf_counter() - started
74
+
75
+ # Metrics from scope manifests + audit reports.
76
+ total_chars = 0
77
+ total_tokens_est = 0
78
+ verified = 0
79
+ for s in fixed_scopes:
80
+ sid = s["scope_id"]
81
+ scope_dir = base_dir / "scratchpads" / sid
82
+ for name in ("alpha_dossier.json", "beta_dossier.json"):
83
+ p = scope_dir / name
84
+ if p.exists():
85
+ total_chars += p.stat().st_size
86
+ audit_path = scope_dir / "audit_report.json"
87
+ if audit_path.exists():
88
+ try:
89
+ summary = json.loads(audit_path.read_text()).get("summary", {})
90
+ verified += int(summary.get("verified_passed", 0))
91
+ except (OSError, json.JSONDecodeError):
92
+ pass
93
+ manifest_path = scope_dir / "manifest.json"
94
+ if manifest_path.exists():
95
+ try:
96
+ telemetry = json.loads(manifest_path.read_text()).get("telemetry", {})
97
+ if "tokens_used" in telemetry:
98
+ total_tokens_est += int(telemetry["tokens_used"])
99
+ except (OSError, json.JSONDecodeError):
100
+ pass
101
+
102
+ if total_tokens_est == 0:
103
+ total_tokens_est = estimate_tokens("x" * total_chars)
104
+
105
+ return {
106
+ "allocation": allocation,
107
+ "wall_clock_s": round(elapsed, 3),
108
+ "verified_claims": verified,
109
+ "chars": total_chars,
110
+ "tokens_estimate": total_tokens_est, # FLAGGED APPROXIMATION (chars/4)
111
+ "claims_per_token": (
112
+ round(verified / total_tokens_est, 6) if total_tokens_est else None
113
+ ),
114
+ }
115
+
116
+
117
+ def main(argv: Optional[list] = None) -> int:
118
+ parser = argparse.ArgumentParser(description="Stream F auction experiment")
119
+ parser.add_argument("--mock", action="store_true", help="Mock mode, zero API cost")
120
+ parser.add_argument("--out-dir", default=None)
121
+ parser.add_argument("--fixture", default=str(FIXTURE))
122
+ args = parser.parse_args(argv)
123
+
124
+ fixture = load_fixture(Path(args.fixture))
125
+ out_dir = Path(args.out_dir) if args.out_dir else (PROJECT_ROOT / ".research" / "experiments")
126
+ out_dir.mkdir(parents=True, exist_ok=True)
127
+
128
+ results = {}
129
+ for policy in ("dag", "auction"):
130
+ print(f"[{policy}] run...", flush=True)
131
+ results[policy] = run_once(
132
+ fixture, out_dir / "work" / f"auction_{policy}", policy, args.mock
133
+ )
134
+
135
+ dag, auction = results["dag"], results["auction"]
136
+ wall_ratio = (
137
+ round(auction["wall_clock_s"] / dag["wall_clock_s"], 4)
138
+ if dag["wall_clock_s"] > 0 else None
139
+ )
140
+ cpt_gain = (
141
+ round((auction["claims_per_token"] - dag["claims_per_token"]) / dag["claims_per_token"], 4)
142
+ if dag["claims_per_token"] else None
143
+ )
144
+
145
+ if args.mock:
146
+ gate = "UNMEASURED_MOCK"
147
+ elif cpt_gain is None:
148
+ gate = "UNMEASURED"
149
+ else:
150
+ gate = "PASS" if cpt_gain >= GATE_BAR_CLAIMS_PER_TOKEN else "FAIL"
151
+
152
+ artifact = {
153
+ "experiment": "auction-f",
154
+ "ran_at": datetime.now(timezone.utc).isoformat(),
155
+ "mock": args.mock,
156
+ "fixture": args.fixture,
157
+ "gate_bar_claims_per_token_gain": GATE_BAR_CLAIMS_PER_TOKEN,
158
+ "results": results,
159
+ "wall_clock_ratio_auction_over_dag": wall_ratio,
160
+ "claims_per_token_gain": cpt_gain,
161
+ "gate_f": gate,
162
+ "token_note": "tokens_estimate is chars/4 — a flagged approximation, not a measurement",
163
+ }
164
+
165
+ stamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ")
166
+ out_path = out_dir / f"auction_{stamp}.json"
167
+ with open(out_path, "w", encoding="utf-8") as f:
168
+ json.dump(artifact, f, indent=2)
169
+
170
+ print(json.dumps({
171
+ "gate_f": gate,
172
+ "claims_per_token_gain": cpt_gain,
173
+ "wall_clock_ratio": wall_ratio,
174
+ "artifact": str(out_path),
175
+ }, indent=2))
176
+ return 0
177
+
178
+
179
+ if __name__ == "__main__":
180
+ sys.exit(main())
@@ -0,0 +1,183 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ IUMBTEMS Universal Harness Adapter builder.
4
+
5
+ Single canonical source of truth: skills/*/SKILL.md + prompts/*.md.
6
+ Since the MCP-canonical migration (Stream A2b), harness mirrors are THIN STUBS,
7
+ not code copies: each target skill dir holds one ~10-line SKILL.md pointing at
8
+ the canonical prose (skills/<skill>/) and the canonical programmatic surface
9
+ (runner/mcp_server.py). Full skill copies were retired deliberately — see
10
+ .research/scratchpads/brainstorm_where-do-we-go-from-here/g1_probe.md.
11
+
12
+ Targets (each gets one stub per canonical skill):
13
+ plugins/antigravity/skills/<skill>/SKILL.md
14
+ plugins/gemini/skills/<skill>/SKILL.md
15
+ plugins/codex/skills/<skill>/SKILL.md
16
+ .agents/skills/<skill>/SKILL.md
17
+
18
+ Also validates every manifest referenced in package.json pi/omp blocks.
19
+
20
+ Usage:
21
+ python3 scripts/build_adapters.py # build all stubs
22
+ python3 scripts/build_adapters.py --check # verify stubs match template (CI)
23
+ python3 scripts/build_adapters.py --clean # remove generated stubs
24
+ """
25
+
26
+ import argparse
27
+ import json
28
+ import shutil
29
+ import sys
30
+ from pathlib import Path
31
+
32
+ PROJECT_ROOT = Path(__file__).resolve().parent.parent
33
+ CANONICAL_SKILLS = [
34
+ "grilling",
35
+ "research_cache",
36
+ "epistemic_search",
37
+ "swarm_config",
38
+ "code_audit",
39
+ "oss_scout",
40
+ "brainstorming",
41
+ ]
42
+
43
+ # Skill -> the MCP tool(s) that now carry its programmatic surface.
44
+ SKILL_TOOLS = {
45
+ "grilling": ["iumbtems_socratic_frontier"],
46
+ "research_cache": ["iumbtems_verify_quote"],
47
+ "epistemic_search": ["brave_web_search / firecrawl_scrape (research MCP servers)"],
48
+ "swarm_config": ["iumbtems_config"],
49
+ "code_audit": ["iumbtems_code_audit"],
50
+ "oss_scout": ["iumbtems_oss_scout"],
51
+ "brainstorming": ["iumbtems_brainstorm"],
52
+ }
53
+
54
+ SKILL_TITLES = {
55
+ "grilling": "Socratic Grilling",
56
+ "research_cache": "Research Cache",
57
+ "epistemic_search": "Epistemic Search",
58
+ "swarm_config": "Swarm Config",
59
+ "code_audit": "Code Audit",
60
+ "oss_scout": "OSS Scout",
61
+ "brainstorming": "Brainstorming",
62
+ }
63
+
64
+ # (target dir relative to root, mode). All targets are stubs since A2b.
65
+ TARGETS = [
66
+ ("plugins/antigravity/skills", "stub"),
67
+ ("plugins/gemini/skills", "stub"),
68
+ ("plugins/codex/skills", "stub"),
69
+ (".agents/skills", "stub"),
70
+ ]
71
+
72
+ STUB_TEMPLATE = """# {title} (thin adapter stub)
73
+
74
+ This file is a POINTER, not the implementation. It exists so harness skill
75
+ discovery finds an entry; the real skill lives in the IUMBTEMS repo.
76
+
77
+ - Canonical prose & scripts: `skills/{skill}/`
78
+ - Canonical programmatic surface: `python3 runner/mcp_server.py` (stdio MCP),
79
+ or one-shot: `python3 runner/mcp_server.py call <tool> '{{...json...}}'`
80
+ - MCP tools for this skill: {tools}
81
+
82
+ Epistemic rules apply regardless of harness: tag claims as
83
+ `[VERIFIED: <hash>]`, `[INFERRED: <reasoning>]`, `[HYPOTHESIS: <test>]`, or
84
+ `[NEGATIVE_KNOWLEDGE: <query>]`. Writes go only to `.research/`.
85
+ """
86
+
87
+
88
+ def render_stub(skill: str) -> str:
89
+ tools = ", ".join(f"`{t}`" for t in SKILL_TOOLS.get(skill, []))
90
+ return STUB_TEMPLATE.format(
91
+ title=SKILL_TITLES.get(skill, skill.replace("_", " ").title()),
92
+ skill=skill,
93
+ tools=tools,
94
+ )
95
+
96
+
97
+ def build():
98
+ count = 0
99
+ for target_rel, mode in TARGETS:
100
+ target_base = PROJECT_ROOT / target_rel
101
+ target_base.mkdir(parents=True, exist_ok=True)
102
+ for skill in CANONICAL_SKILLS:
103
+ src = PROJECT_ROOT / "skills" / skill
104
+ if not src.exists():
105
+ print(f"[WARN] canonical skill missing: skills/{skill}", file=sys.stderr)
106
+ continue
107
+ if not (src / "SKILL.md").exists():
108
+ print(f"[WARN] skills/{skill}/SKILL.md missing", file=sys.stderr)
109
+ continue
110
+ dst = target_base / skill
111
+ if dst.exists():
112
+ shutil.rmtree(dst)
113
+ dst.mkdir(parents=True, exist_ok=True)
114
+ (dst / "SKILL.md").write_text(render_stub(skill), encoding="utf-8")
115
+ count += 1
116
+ print(f"✅ Built {count} thin skill stubs across {len(TARGETS)} targets.")
117
+ return 0
118
+
119
+
120
+ def check():
121
+ """Verify stubs match the generated template (not the skill bodies)."""
122
+ errors = []
123
+ for target_rel, mode in TARGETS:
124
+ for skill in CANONICAL_SKILLS:
125
+ dst = PROJECT_ROOT / target_rel / skill / "SKILL.md"
126
+ if not dst.exists():
127
+ errors.append(f"MISSING stub: {target_rel}/{skill}/SKILL.md")
128
+ continue
129
+ if dst.read_text(encoding="utf-8") != render_stub(skill):
130
+ errors.append(f"OUT OF SYNC: {target_rel}/{skill}/SKILL.md")
131
+ # Stubs must not drag code bodies back in.
132
+ stray = [
133
+ p.name
134
+ for p in (PROJECT_ROOT / target_rel / skill).iterdir()
135
+ if p.name != "SKILL.md"
136
+ ]
137
+ if stray:
138
+ errors.append(f"STRAY FILES in {target_rel}/{skill}: {sorted(stray)}")
139
+ # Validate package.json pi/omp blocks resolve
140
+ pkg = json.loads((PROJECT_ROOT / "package.json").read_text())
141
+ for key in ("pi", "omp"):
142
+ block = pkg.get(key, {})
143
+ for kind in ("skills", "extensions", "prompts"):
144
+ for p in block.get(kind, []):
145
+ # glob patterns allowed in prompts
146
+ if "*" in p:
147
+ if not list(PROJECT_ROOT.glob(p.lstrip("./"))):
148
+ errors.append(
149
+ f"package.json {key}.{kind} glob matches nothing: {p}"
150
+ )
151
+ elif not (PROJECT_ROOT / p.lstrip("./")).exists():
152
+ errors.append(f"package.json {key}.{kind} missing: {p}")
153
+ if errors:
154
+ print("❌ Adapter check failed:")
155
+ for e in errors:
156
+ print(f" - {e}")
157
+ return 1
158
+ print("✅ All adapter stubs in sync with template; package.json pi/omp blocks resolve.")
159
+ return 0
160
+
161
+
162
+ def clean():
163
+ for target_rel, _ in TARGETS:
164
+ base = PROJECT_ROOT / target_rel
165
+ if base.exists():
166
+ shutil.rmtree(base)
167
+ print(f"Removed {target_rel}")
168
+
169
+
170
+ def main():
171
+ ap = argparse.ArgumentParser(description="IUMBTEMS adapter builder")
172
+ ap.add_argument("--check", action="store_true")
173
+ ap.add_argument("--clean", action="store_true")
174
+ args = ap.parse_args()
175
+ if args.clean:
176
+ return clean()
177
+ if args.check:
178
+ return check()
179
+ return build()
180
+
181
+
182
+ if __name__ == "__main__":
183
+ sys.exit(main())
@@ -0,0 +1,184 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ Stream E / Gate G2: cross-family adversary divergence experiment.
4
+
5
+ Probe: [HYPOTHESIS: cross-family Alpha/Beta pairs raise mean Divergence
6
+ Score by >= 0.15 vs same-family pairs across 5 fixed objectives]
7
+
8
+ Runs each fixture objective twice — once with a same-family config, once with
9
+ a cross-family config — and emits a delta artifact under
10
+ .research/experiments/. LIVE MODE SPENDS MONEY: it drives real backends.
11
+
12
+ Zero-dep: stdlib + the existing SwarmRunner. Token/cost accounting is
13
+ best-effort (chars/4) and reported as an estimate, not a measurement.
14
+
15
+ Usage:
16
+ # Dry-run (mock mode, zero cost) — proves the harness, not the hypothesis
17
+ python3 scripts/divergence_experiment.py --mock
18
+
19
+ # Live (requires configured backends and spend approval)
20
+ python3 scripts/divergence_experiment.py \
21
+ --beta-backend ollama run qwen3 --model-alpha claude-opus-5
22
+ """
23
+
24
+ import argparse
25
+ import json
26
+ import sys
27
+ import time
28
+ from datetime import datetime, timezone
29
+ from pathlib import Path
30
+ from typing import Any, Dict, List, Optional
31
+
32
+ PROJECT_ROOT = Path(__file__).resolve().parent.parent
33
+ if str(PROJECT_ROOT) not in sys.path:
34
+ sys.path.insert(0, str(PROJECT_ROOT))
35
+
36
+ from runner.research_swarm import SwarmRunner # noqa: E402
37
+
38
+ FIXTURE = PROJECT_ROOT / "runner" / "tests" / "fixtures" / "divergence_objectives.json"
39
+ GATE_BAR = 0.15
40
+
41
+
42
+ def load_objectives(path: Path = FIXTURE) -> List[Dict[str, Any]]:
43
+ with open(path, "r", encoding="utf-8") as f:
44
+ data = json.load(f)
45
+ return data["objectives"]
46
+
47
+
48
+ def estimate_tokens(text: str) -> int:
49
+ """Rough token estimate (chars/4). FLAGGED APPROXIMATION, not a measurement."""
50
+ return max(1, len(text) // 4)
51
+
52
+
53
+ def run_one(
54
+ objective: str,
55
+ base_dir: Path,
56
+ agent_overrides: Optional[Dict[str, Dict[str, Any]]],
57
+ mock: bool,
58
+ ) -> Dict[str, Any]:
59
+ """Run one swarm under one config and pull divergence out of the audit."""
60
+ runner = SwarmRunner(
61
+ base_dir=base_dir,
62
+ mock_mode=mock,
63
+ mode="research",
64
+ agent_overrides=agent_overrides,
65
+ )
66
+ started = time.perf_counter()
67
+ # Suppress the swarm's chatty stdout; the artifact is what matters.
68
+ import contextlib, io
69
+
70
+ buf = io.StringIO()
71
+ with contextlib.redirect_stdout(buf):
72
+ runner.run_swarm(objective)
73
+ elapsed = time.perf_counter() - started
74
+
75
+ report_path = base_dir / "final_synthesis.md"
76
+ divergence = None
77
+ verified = None
78
+ audit_paths = sorted((base_dir / "scratchpads").glob("*/audit_report.json"))
79
+ for ap in audit_paths:
80
+ try:
81
+ with open(ap, "r", encoding="utf-8") as f:
82
+ audit = json.load(f)
83
+ summary = audit.get("summary", {})
84
+ if summary.get("divergence_score") is not None:
85
+ divergence = summary["divergence_score"]
86
+ verified = summary.get("verified_passed")
87
+ except (OSError, json.JSONDecodeError):
88
+ continue
89
+
90
+ log = buf.getvalue()
91
+ return {
92
+ "divergence_score": divergence,
93
+ "verified_passed": verified,
94
+ "wall_clock_s": round(elapsed, 3),
95
+ "tokens_estimate": estimate_tokens(log), # FLAGGED APPROXIMATION
96
+ "report": str(report_path) if report_path.exists() else None,
97
+ }
98
+
99
+
100
+ def main(argv: Optional[List[str]] = None) -> int:
101
+ parser = argparse.ArgumentParser(description="Stream E divergence experiment (G2)")
102
+ parser.add_argument("--mock", action="store_true", help="Mock mode, zero API cost")
103
+ parser.add_argument("--out-dir", default=None, help="Artifact directory")
104
+ parser.add_argument("--model-alpha", default=None)
105
+ parser.add_argument("--model-beta", default=None)
106
+ parser.add_argument("--beta-backend", nargs="+", default=None)
107
+ parser.add_argument("--fixture", default=str(FIXTURE))
108
+ args = parser.parse_args(argv)
109
+
110
+ objectives = load_objectives(Path(args.fixture))
111
+ out_dir = Path(args.out_dir) if args.out_dir else (PROJECT_ROOT / ".research" / "experiments")
112
+ out_dir.mkdir(parents=True, exist_ok=True)
113
+
114
+ same_family: Dict[str, Dict[str, Any]] = {}
115
+ cross_family: Dict[str, Dict[str, Any]] = {}
116
+ if args.model_alpha:
117
+ same_family.setdefault("alpha", {})["model"] = args.model_alpha
118
+ same_family.setdefault("beta", {})["model"] = args.model_alpha
119
+ if args.model_beta:
120
+ cross_family.setdefault("beta", {})["model"] = args.model_beta
121
+ if args.beta_backend:
122
+ cross_family.setdefault("beta", {})["backend"] = args.beta_backend
123
+
124
+ results = []
125
+ for obj in objectives:
126
+ oid = obj["id"]
127
+ print(f"[{oid}] same-family run...", flush=True)
128
+ same = run_one(
129
+ obj["objective"],
130
+ out_dir / "work" / oid / "same",
131
+ same_family or None,
132
+ args.mock,
133
+ )
134
+ print(f"[{oid}] cross-family run...", flush=True)
135
+ cross = run_one(
136
+ obj["objective"],
137
+ out_dir / "work" / oid / "cross",
138
+ cross_family or None,
139
+ args.mock,
140
+ )
141
+ d_same = same["divergence_score"]
142
+ d_cross = cross["divergence_score"]
143
+ delta = (
144
+ round(d_cross - d_same, 4)
145
+ if (d_same is not None and d_cross is not None)
146
+ else None
147
+ )
148
+ results.append({"id": oid, "objective": obj["objective"], "same": same, "cross": cross, "delta_divergence": delta})
149
+
150
+ deltas = [r["delta_divergence"] for r in results if r["delta_divergence"] is not None]
151
+ mean_delta = round(sum(deltas) / len(deltas), 4) if deltas else None
152
+
153
+ # A mock run cannot falsify G2: both arms drive the same canned backend,
154
+ # so a 0.0 delta is a property of the fixture, not of model diversity.
155
+ if args.mock:
156
+ gate_status = "UNMEASURED_MOCK"
157
+ elif mean_delta is None:
158
+ gate_status = "UNMEASURED"
159
+ else:
160
+ gate_status = "PASS" if mean_delta >= GATE_BAR else "FAIL"
161
+
162
+ artifact = {
163
+ "experiment": "divergence-g2",
164
+ "ran_at": datetime.now(timezone.utc).isoformat(),
165
+ "mock": args.mock,
166
+ "fixture": args.fixture,
167
+ "gate_bar_delta": GATE_BAR,
168
+ "results": results,
169
+ "mean_delta_divergence": mean_delta,
170
+ "gate_g2": gate_status,
171
+ "token_note": "tokens_estimate is chars/4 — a flagged approximation, not a measurement",
172
+ }
173
+
174
+ stamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ")
175
+ out_path = out_dir / f"divergence_{stamp}.json"
176
+ with open(out_path, "w", encoding="utf-8") as f:
177
+ json.dump(artifact, f, indent=2)
178
+
179
+ print(json.dumps({"mean_delta": mean_delta, "gate_g2": artifact["gate_g2"], "artifact": str(out_path)}, indent=2))
180
+ return 0
181
+
182
+
183
+ if __name__ == "__main__":
184
+ sys.exit(main())
@@ -0,0 +1,106 @@
1
+ ---
2
+ name: brainstorming
3
+ description: Lateral creative brainstorming and divergent ideation skill. Use when exploring what-if features, lateral architectures, speculative product directions, or when user invokes /brainstorming. Generates novel feature vectors, paradigm inversions, and falsifiable spike hypotheses instead of bug fixes.
4
+ ---
5
+
6
+ # Creative Brainstorming & Lateral Ideation Engine
7
+
8
+ Trigger with `/brainstorming <ambiguous-prompt>` (alias `/brainstorm`).
9
+ This skill inverts the default coding-assistant bias (bug fixes, incremental
10
+ refactors) and forces divergent, speculative, lateral thinking.
11
+
12
+ ## 1. Invocation
13
+
14
+ ```bash
15
+ # Interactive / single-pass brainstorm (no LLM tokens required for scaffolding)
16
+ python3 skills/brainstorming/scripts/brainstorm.py --objective "Where do we go from here?" --show-context
17
+
18
+ # Full dialectic brainstorm swarm (mock = zero token cost)
19
+ python3 runner/research_swarm.py --mode brainstorm --objective "What-if gameplay mechanics for HarborTown" --mock-claude
20
+
21
+ # Live swarm (invokes Claude Code headless sessions)
22
+ python3 runner/research_swarm.py --mode brainstorm --objective "<objective>"
23
+
24
+ # Via CLI
25
+ iumbtems brainstorm "Where do we go from here?"
26
+ iumbtems brainstorm "What-if mechanics for HarborTown" --mock-claude
27
+ ```
28
+
29
+ In Pi / OMP: `/brainstorming <prompt>`. In OpenCode: `iumbtems_brainstorm` tool.
30
+ In Gemini / AntiGravity: skill auto-activates on speculative intent.
31
+ In Codex: `$brainstorming <prompt>`.
32
+
33
+ ## 2. Dynamic Context Ingestion (read-only, capped)
34
+
35
+ Before ideating, the agent MUST ingest (never exceed ~30% of context budget):
36
+
37
+ 1. Project tree: `glob` depth 3 + `README.md`, `AGENTS.md`/`GEMINI.md`/`SYSTEM.md` if present.
38
+ 2. Architecture docs: `docs/SYSTEM_ARCHITECTURE.md`, `docs/*`, `prompts/*.md` domain hints.
39
+ 3. Recent history: `git log --oneline -20`, `git status --short`.
40
+ 4. Open loops: `grep -r "TODO|FIXME|HACK|XXX" --include="*.py" --include="*.ts" --include="*.js" .`
41
+ 5. Stack fingerprint: `package.json`, `pyproject.toml`, `Cargo.toml`, `go.mod`, `requirements*.txt`.
42
+
43
+ Record the result as `domain_model.json`:
44
+
45
+ ```json
46
+ {
47
+ "domain": "<game mechanics|data pipeline|compiler toolchain|multi-agent runtime|...>",
48
+ "entities": ["<core entity>"],
49
+ "constraints": ["<hard constraint>"],
50
+ "stack": ["<detected dependency>"],
51
+ "open_loops": ["<TODO>"]
52
+ }
53
+ ```
54
+
55
+ If context budget is < 40% free, fall back to single-pass mode: skip git
56
+ history and full tree, use only README + top-level listing.
57
+
58
+ ## 3. Dialectical Divergence (mandatory)
59
+
60
+ Run Thesis / Radical Antithesis / Synthesis. Do NOT produce bug-fix lists.
61
+
62
+ ### Thesis — Wild Proponent
63
+ Propose the most ambitious affirmative vision: 10x features, new mechanics,
64
+ new workflows. No feasibility filter.
65
+
66
+ ### Radical Antithesis — Paradigm Inverter
67
+ Invert every Thesis premise:
68
+ - *Inversion*: What if the core objective is obsolete? What replaces it?
69
+ - *Subtraction*: What if we remove the most "essential" component?
70
+ - *Adjacent domains*: What would biology / compilers / markets / games do here?
71
+ - *Cutting-edge OSS*: What new dependency, protocol, or model unlocks a shortcut?
72
+ - *Counter-intuition*: What deliberately "wrong" optimization wins?
73
+
74
+ ### Synthesis — Portfolio Builder
75
+ Output exactly:
76
+
77
+ 1. **Novel Feature Vectors (3-5)**: high-impact "what-if" mechanics or workflows.
78
+ Each: name, one-line pitch, why now, falsification probe.
79
+ 2. **Lateral Architectural Moves (2-3)**: alternative paradigms, OSS swaps,
80
+ counter-intuitive optimizations. Each: current vs. lateral, trade-off, migration spike.
81
+ 3. **Experimental Hypotheses (2-3)**: rapid spikes (< 1 day each). Each MUST carry
82
+ `[HYPOTHESIS: <measurable falsification criterion>]`.
83
+
84
+ ## 4. Execution Pipeline & Calibration
85
+
86
+ | Parameter | Brainstorm value | Research/audit value |
87
+ |---|---|---|
88
+ | `temperature` | 1.0 – 1.2 | 0.2 – 0.4 |
89
+ | `top_p` | 0.92 | 0.9 |
90
+ | `max_iterations` | 1 – 2 | 2 – 4 |
91
+ | `divergence` | rewarded (target > 0.6) | audited (threshold 0.75) |
92
+ | `epistemic bar` | `[HYPOTHESIS]` required, `[VERIFIED]` optional | `[VERIFIED]` required |
93
+
94
+ Topology: isolated subagent by default. With `--swarm`, dispatch via
95
+ `runner/research_swarm.py --mode brainstorm` (Alpha=warm wild proponent,
96
+ Beta=radical inverter, Auditor=divergence-rewarding synthesizer).
97
+ Write output ONLY to `.research/brainstorm_<timestamp>.md` plus
98
+ `scratchpads/brainstorm_<slug>/domain_model.json`.
99
+
100
+ ## 5. Anti-Patterns (hard bans)
101
+
102
+ - No bug-fix or incremental-refactor lists as "ideas".
103
+ - No sycophantic convergence ("all options are great").
104
+ - No parametric citations presented as `[VERIFIED]` — speculation MUST be
105
+ tagged `[HYPOTHESIS: <test>]` or `[INFERRED: <parents>]`.
106
+ - No writes outside `.research/`.
@@ -0,0 +1 @@
1
+ # AgentSkills compatibility marker