@heretek-ai/epistemic-swarm 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +14 -0
- package/.claude-plugin/plugin.json +58 -0
- package/LICENSE +126 -0
- package/README.md +130 -0
- package/bin/cli.js +119 -0
- package/config/claude-settings-patch.json +10 -0
- package/config/docker-compose.infra.yml +23 -0
- package/config/mcp-research-servers.json +26 -0
- package/config/searxng_mcp.py +133 -0
- package/install.sh +102 -0
- package/package.json +52 -0
- package/prompts/agent_alpha_thesis.md +70 -0
- package/prompts/agent_beta_antithesis.md +78 -0
- package/prompts/base_epistemic_system.md +50 -0
- package/prompts/epistemic_auditor.md +74 -0
- package/prompts/orchestrator.md +70 -0
- package/runner/__init__.py +0 -0
- package/runner/__pycache__/__init__.cpython-314.pyc +0 -0
- package/runner/__pycache__/auditor_engine.cpython-314.pyc +0 -0
- package/runner/__pycache__/research_swarm.cpython-314.pyc +0 -0
- package/runner/__pycache__/state_machine.cpython-314.pyc +0 -0
- package/runner/auditor_engine.py +222 -0
- package/runner/research_swarm.py +337 -0
- package/runner/state_machine.py +192 -0
- package/runner/tests/__pycache__/test_swarm.cpython-314.pyc +0 -0
- package/runner/tests/test_swarm.py +178 -0
- package/skills/grilling/SKILL.md +48 -0
- package/skills/grilling/__init__.py +0 -0
- package/skills/grilling/socratic_tree.py +148 -0
- package/skills/research-cache/SKILL.md +36 -0
- package/skills/research-cache/__init__.py +0 -0
- package/skills/research-cache/__pycache__/__init__.cpython-314.pyc +0 -0
- package/skills/research-cache/__pycache__/hasher.cpython-314.pyc +0 -0
- package/skills/research-cache/hasher.py +195 -0
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Filesystem IPC Protocol & State Machine for Epistemic Swarm.
|
|
4
|
+
Manages .research/ hierarchy, session state, scope DAG, and dossier serialization.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import os
|
|
8
|
+
import json
|
|
9
|
+
import threading
|
|
10
|
+
import tempfile
|
|
11
|
+
from enum import Enum
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from datetime import datetime, timezone
|
|
14
|
+
from typing import Dict, Any, List, Optional
|
|
15
|
+
|
|
16
|
+
class SessionStatus(str, Enum):
|
|
17
|
+
INITIALIZING = "INITIALIZING"
|
|
18
|
+
FRONTIER_SETTLING = "FRONTIER_SETTLING"
|
|
19
|
+
ORCHESTRATING = "ORCHESTRATING"
|
|
20
|
+
SWARM_DISPATCHED = "SWARM_DISPATCHED"
|
|
21
|
+
AUDITING = "AUDITING"
|
|
22
|
+
COMPLETED = "COMPLETED"
|
|
23
|
+
FAILED = "FAILED"
|
|
24
|
+
|
|
25
|
+
class ScopeStatus(str, Enum):
|
|
26
|
+
PENDING = "PENDING"
|
|
27
|
+
ALPHA_RUNNING = "ALPHA_RUNNING"
|
|
28
|
+
BETA_RUNNING = "BETA_RUNNING"
|
|
29
|
+
RUNNING_PARALLEL = "RUNNING_PARALLEL"
|
|
30
|
+
ALPHA_COMPLETE = "ALPHA_COMPLETE"
|
|
31
|
+
BETA_COMPLETE = "BETA_COMPLETE"
|
|
32
|
+
DOSSIERS_READY = "DOSSIERS_READY"
|
|
33
|
+
AUDITING = "AUDITING"
|
|
34
|
+
COMPLETE = "COMPLETE"
|
|
35
|
+
FAILED = "FAILED"
|
|
36
|
+
|
|
37
|
+
class ResearchStateMachine:
|
|
38
|
+
def __init__(self, base_dir: Optional[Path] = None):
|
|
39
|
+
self.base_dir = base_dir or Path(".research")
|
|
40
|
+
self.scratchpads_dir = self.base_dir / "scratchpads"
|
|
41
|
+
self.sources_dir = self.base_dir / "sources"
|
|
42
|
+
self.manifest_file = self.base_dir / "manifest.json"
|
|
43
|
+
self._lock = threading.Lock()
|
|
44
|
+
|
|
45
|
+
# Ensure directories exist
|
|
46
|
+
self.scratchpads_dir.mkdir(parents=True, exist_ok=True)
|
|
47
|
+
self.sources_dir.mkdir(parents=True, exist_ok=True)
|
|
48
|
+
|
|
49
|
+
def init_session(self, objective: str, session_id: Optional[str] = None) -> Dict[str, Any]:
|
|
50
|
+
"""Initialize or reset a research session manifest."""
|
|
51
|
+
sid = session_id or f"session-{datetime.now(timezone.utc).strftime('%Y%m%d-%H%M%S')}"
|
|
52
|
+
manifest = {
|
|
53
|
+
"session_id": sid,
|
|
54
|
+
"objective": objective,
|
|
55
|
+
"status": SessionStatus.INITIALIZING.value,
|
|
56
|
+
"created_at": datetime.now(timezone.utc).isoformat(),
|
|
57
|
+
"updated_at": datetime.now(timezone.utc).isoformat(),
|
|
58
|
+
"scopes": [],
|
|
59
|
+
"telemetry": {
|
|
60
|
+
"total_scopes": 0,
|
|
61
|
+
"completed_scopes": 0,
|
|
62
|
+
"total_claims_audited": 0,
|
|
63
|
+
"total_verified_claims": 0,
|
|
64
|
+
"total_rejected_claims": 0,
|
|
65
|
+
"mean_epistemic_score": 0.0,
|
|
66
|
+
"mean_divergence_score": 0.0
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
self.save_global_manifest(manifest)
|
|
70
|
+
return manifest
|
|
71
|
+
|
|
72
|
+
def load_global_manifest(self) -> Dict[str, Any]:
|
|
73
|
+
with self._lock:
|
|
74
|
+
if not self.manifest_file.exists():
|
|
75
|
+
raise FileNotFoundError(f"Global manifest not found at {self.manifest_file}")
|
|
76
|
+
with open(self.manifest_file, "r", encoding="utf-8") as f:
|
|
77
|
+
return json.load(f)
|
|
78
|
+
|
|
79
|
+
def save_global_manifest(self, manifest: Dict[str, Any]):
|
|
80
|
+
with self._lock:
|
|
81
|
+
manifest["updated_at"] = datetime.now(timezone.utc).isoformat()
|
|
82
|
+
temp_path = self.manifest_file.with_suffix(".tmp")
|
|
83
|
+
with open(temp_path, "w", encoding="utf-8") as f:
|
|
84
|
+
json.dump(manifest, f, indent=2)
|
|
85
|
+
os.replace(temp_path, self.manifest_file)
|
|
86
|
+
|
|
87
|
+
def update_session_status(self, status: SessionStatus):
|
|
88
|
+
manifest = self.load_global_manifest()
|
|
89
|
+
manifest["status"] = status.value
|
|
90
|
+
self.save_global_manifest(manifest)
|
|
91
|
+
|
|
92
|
+
def set_scopes(self, scopes: List[Dict[str, Any]]):
|
|
93
|
+
"""Set scopes decomposed by orchestrator and prepare scratchpads."""
|
|
94
|
+
manifest = self.load_global_manifest()
|
|
95
|
+
manifest["scopes"] = scopes
|
|
96
|
+
manifest["telemetry"]["total_scopes"] = len(scopes)
|
|
97
|
+
self.save_global_manifest(manifest)
|
|
98
|
+
|
|
99
|
+
for scope in scopes:
|
|
100
|
+
scope_id = scope["scope_id"]
|
|
101
|
+
scope_dir = self.scratchpads_dir / scope_id
|
|
102
|
+
scope_dir.mkdir(parents=True, exist_ok=True)
|
|
103
|
+
|
|
104
|
+
scope_manifest = {
|
|
105
|
+
"scope_id": scope_id,
|
|
106
|
+
"title": scope.get("title", ""),
|
|
107
|
+
"objective": scope.get("objective", ""),
|
|
108
|
+
"dependencies": scope.get("dependencies", []),
|
|
109
|
+
"status": ScopeStatus.PENDING.value,
|
|
110
|
+
"alpha_completed": False,
|
|
111
|
+
"beta_completed": False,
|
|
112
|
+
"audit_completed": False,
|
|
113
|
+
"created_at": datetime.now(timezone.utc).isoformat()
|
|
114
|
+
}
|
|
115
|
+
temp_scope_file = (scope_dir / "manifest.json").with_suffix(".tmp")
|
|
116
|
+
with open(temp_scope_file, "w", encoding="utf-8") as f:
|
|
117
|
+
json.dump(scope_manifest, f, indent=2)
|
|
118
|
+
os.replace(temp_scope_file, scope_dir / "manifest.json")
|
|
119
|
+
|
|
120
|
+
def get_scope_dir(self, scope_id: str) -> Path:
|
|
121
|
+
return self.scratchpads_dir / scope_id
|
|
122
|
+
|
|
123
|
+
def load_scope_manifest(self, scope_id: str) -> Dict[str, Any]:
|
|
124
|
+
with self._lock:
|
|
125
|
+
scope_manifest_file = self.get_scope_dir(scope_id) / "manifest.json"
|
|
126
|
+
if not scope_manifest_file.exists():
|
|
127
|
+
raise FileNotFoundError(f"Scope manifest not found for {scope_id}")
|
|
128
|
+
with open(scope_manifest_file, "r", encoding="utf-8") as f:
|
|
129
|
+
return json.load(f)
|
|
130
|
+
|
|
131
|
+
def save_scope_manifest(self, scope_id: str, manifest: Dict[str, Any]):
|
|
132
|
+
with self._lock:
|
|
133
|
+
scope_manifest_file = self.get_scope_dir(scope_id) / "manifest.json"
|
|
134
|
+
temp_path = scope_manifest_file.with_suffix(".tmp")
|
|
135
|
+
with open(temp_path, "w", encoding="utf-8") as f:
|
|
136
|
+
json.dump(manifest, f, indent=2)
|
|
137
|
+
os.replace(temp_path, scope_manifest_file)
|
|
138
|
+
|
|
139
|
+
def update_scope_status(self, scope_id: str, status: ScopeStatus):
|
|
140
|
+
sm = self.load_scope_manifest(scope_id)
|
|
141
|
+
sm["status"] = status.value
|
|
142
|
+
self.save_scope_manifest(scope_id, sm)
|
|
143
|
+
|
|
144
|
+
def record_agent_completion(self, scope_id: str, agent_type: str, dossier_data: Dict[str, Any]):
|
|
145
|
+
"""Records dossier from Alpha or Beta and advances scope state machine."""
|
|
146
|
+
scope_dir = self.get_scope_dir(scope_id)
|
|
147
|
+
|
|
148
|
+
if agent_type.lower() in ["alpha", "thesis", "proponent"]:
|
|
149
|
+
filename = "alpha_dossier.json"
|
|
150
|
+
is_alpha = True
|
|
151
|
+
elif agent_type.lower() in ["beta", "antithesis", "adversary", "red_team"]:
|
|
152
|
+
filename = "beta_dossier.json"
|
|
153
|
+
is_alpha = False
|
|
154
|
+
else:
|
|
155
|
+
raise ValueError(f"Unknown agent type: {agent_type}")
|
|
156
|
+
|
|
157
|
+
temp_dossier = (scope_dir / filename).with_suffix(".tmp")
|
|
158
|
+
with open(temp_dossier, "w", encoding="utf-8") as f:
|
|
159
|
+
json.dump(dossier_data, f, indent=2)
|
|
160
|
+
os.replace(temp_dossier, scope_dir / filename)
|
|
161
|
+
|
|
162
|
+
sm = self.load_scope_manifest(scope_id)
|
|
163
|
+
if is_alpha:
|
|
164
|
+
sm["alpha_completed"] = True
|
|
165
|
+
else:
|
|
166
|
+
sm["beta_completed"] = True
|
|
167
|
+
|
|
168
|
+
if sm.get("alpha_completed") and sm.get("beta_completed"):
|
|
169
|
+
sm["status"] = ScopeStatus.DOSSIERS_READY.value
|
|
170
|
+
elif sm.get("alpha_completed"):
|
|
171
|
+
sm["status"] = ScopeStatus.ALPHA_COMPLETE.value
|
|
172
|
+
elif sm.get("beta_completed"):
|
|
173
|
+
sm["status"] = ScopeStatus.BETA_COMPLETE.value
|
|
174
|
+
|
|
175
|
+
self.save_scope_manifest(scope_id, sm)
|
|
176
|
+
|
|
177
|
+
def get_ready_scopes(self) -> List[Dict[str, Any]]:
|
|
178
|
+
"""Return scopes whose dependencies are completed and status is PENDING."""
|
|
179
|
+
manifest = self.load_global_manifest()
|
|
180
|
+
ready = []
|
|
181
|
+
completed_scope_ids = {
|
|
182
|
+
s["scope_id"] for s in manifest["scopes"]
|
|
183
|
+
if self.load_scope_manifest(s["scope_id"]).get("status") == ScopeStatus.COMPLETE.value
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
for scope in manifest["scopes"]:
|
|
187
|
+
sm = self.load_scope_manifest(scope["scope_id"])
|
|
188
|
+
if sm["status"] == ScopeStatus.PENDING.value:
|
|
189
|
+
deps = set(scope.get("dependencies", []))
|
|
190
|
+
if deps.issubset(completed_scope_ids):
|
|
191
|
+
ready.append(scope)
|
|
192
|
+
return ready
|
|
Binary file
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Unit and integration test suite for Epistemic Swarm.
|
|
4
|
+
Tests source hashing, state machine IPC transitions, auditor engine, and swarm mock dispatch.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import sys
|
|
8
|
+
import shutil
|
|
9
|
+
import tempfile
|
|
10
|
+
import unittest
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
# Add project root to sys.path
|
|
14
|
+
PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
|
|
15
|
+
sys.path.insert(0, str(PROJECT_ROOT))
|
|
16
|
+
|
|
17
|
+
from skills.research_cache.hasher import SourceHasher
|
|
18
|
+
from runner.state_machine import ResearchStateMachine, SessionStatus, ScopeStatus
|
|
19
|
+
from runner.auditor_engine import EpistemicAuditorEngine
|
|
20
|
+
from runner.research_swarm import SwarmRunner
|
|
21
|
+
|
|
22
|
+
class TestEpistemicSwarm(unittest.TestCase):
|
|
23
|
+
def setUp(self):
|
|
24
|
+
self.test_dir = Path(tempfile.mkdtemp(prefix="epistemic_test_"))
|
|
25
|
+
self.hasher = SourceHasher(base_dir=self.test_dir)
|
|
26
|
+
self.state_machine = ResearchStateMachine(base_dir=self.test_dir)
|
|
27
|
+
self.auditor = EpistemicAuditorEngine(base_dir=self.test_dir)
|
|
28
|
+
|
|
29
|
+
def tearDown(self):
|
|
30
|
+
shutil.rmtree(self.test_dir, ignore_errors=True)
|
|
31
|
+
|
|
32
|
+
def test_source_hasher_and_quote_verification(self):
|
|
33
|
+
content = """# Deep Learning Scaling Laws
|
|
34
|
+
Empirical measurements show that compute-optimal models scale loss as L(N) = (N_c / N)^alpha_N.
|
|
35
|
+
In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
|
|
36
|
+
"""
|
|
37
|
+
shash = self.hasher.store_source(
|
|
38
|
+
url="https://arxiv.org/abs/2203.15556",
|
|
39
|
+
content=content,
|
|
40
|
+
title="Chinchilla Scaling Laws"
|
|
41
|
+
)
|
|
42
|
+
self.assertTrue(len(shash) == 64)
|
|
43
|
+
|
|
44
|
+
# 1. Exact quote match
|
|
45
|
+
verified, conf, msg = self.hasher.verify_quote(
|
|
46
|
+
shash, "the 70B parameter model was trained on 15.0 trillion tokens."
|
|
47
|
+
)
|
|
48
|
+
self.assertTrue(verified)
|
|
49
|
+
self.assertGreaterEqual(conf, 0.95)
|
|
50
|
+
|
|
51
|
+
# 2. Normalized whitespace match
|
|
52
|
+
verified_norm, conf_norm, _ = self.hasher.verify_quote(
|
|
53
|
+
shash, "the 70B parameter model was\ntrained on 15.0 trillion tokens."
|
|
54
|
+
)
|
|
55
|
+
self.assertTrue(verified_norm)
|
|
56
|
+
|
|
57
|
+
# 3. Fabricated quote rejection
|
|
58
|
+
verified_fake, conf_fake, _ = self.hasher.verify_quote(
|
|
59
|
+
shash, "the 70B parameter model was trained on 500 quadrillion tokens by aliens."
|
|
60
|
+
)
|
|
61
|
+
self.assertFalse(verified_fake)
|
|
62
|
+
self.assertLess(conf_fake, 0.8)
|
|
63
|
+
|
|
64
|
+
def test_state_machine_transitions_and_dag(self):
|
|
65
|
+
self.state_machine.init_session("Evaluate rollup throughput")
|
|
66
|
+
scopes = [
|
|
67
|
+
{
|
|
68
|
+
"scope_id": "scope_01_prover",
|
|
69
|
+
"title": "Prover Benchmarks",
|
|
70
|
+
"dependencies": []
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
"scope_id": "scope_02_recursion",
|
|
74
|
+
"title": "Recursive Verification",
|
|
75
|
+
"dependencies": ["scope_01_prover"]
|
|
76
|
+
}
|
|
77
|
+
]
|
|
78
|
+
self.state_machine.set_scopes(scopes)
|
|
79
|
+
|
|
80
|
+
# Initially, only scope_01 should be ready
|
|
81
|
+
ready = self.state_machine.get_ready_scopes()
|
|
82
|
+
self.assertEqual(len(ready), 1)
|
|
83
|
+
self.assertEqual(ready[0]["scope_id"], "scope_01_prover")
|
|
84
|
+
|
|
85
|
+
# Mark scope_01 complete
|
|
86
|
+
self.state_machine.update_scope_status("scope_01_prover", ScopeStatus.COMPLETE)
|
|
87
|
+
|
|
88
|
+
# Now scope_02 should be ready
|
|
89
|
+
ready_after = self.state_machine.get_ready_scopes()
|
|
90
|
+
self.assertEqual(len(ready_after), 1)
|
|
91
|
+
self.assertEqual(ready_after[0]["scope_id"], "scope_02_recursion")
|
|
92
|
+
|
|
93
|
+
def test_epistemic_auditor_scoring_and_downgrade(self):
|
|
94
|
+
# 1. Seed cached source
|
|
95
|
+
shash = self.hasher.store_source(
|
|
96
|
+
url="https://benchmark.org/zk",
|
|
97
|
+
content="FPGA prover executes Poseidon in 184ms.",
|
|
98
|
+
title="ZK Benchmarks"
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
# 2. Initialize scope
|
|
102
|
+
self.state_machine.init_session("Test Objective")
|
|
103
|
+
self.state_machine.set_scopes([{"scope_id": "scope_test", "dependencies": []}])
|
|
104
|
+
|
|
105
|
+
# 3. Create Alpha Dossier with 1 verified and 1 fake quote
|
|
106
|
+
alpha_dossier = {
|
|
107
|
+
"agent": "Agent Alpha",
|
|
108
|
+
"scope_id": "scope_test",
|
|
109
|
+
"affirmative_claims": [
|
|
110
|
+
{
|
|
111
|
+
"claim_id": "A1",
|
|
112
|
+
"tag": "VERIFIED",
|
|
113
|
+
"statement": "Poseidon prover executes in 184ms",
|
|
114
|
+
"source_hash": shash,
|
|
115
|
+
"verbatim_quote": "FPGA prover executes Poseidon in 184ms."
|
|
116
|
+
},
|
|
117
|
+
{
|
|
118
|
+
"claim_id": "A2",
|
|
119
|
+
"tag": "VERIFIED",
|
|
120
|
+
"statement": "Hallucinated claim that does not exist in source",
|
|
121
|
+
"source_hash": shash,
|
|
122
|
+
"verbatim_quote": "This string does not exist anywhere in the text."
|
|
123
|
+
}
|
|
124
|
+
],
|
|
125
|
+
"negative_knowledge": [{"query": "q1", "finding": "None"}]
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
# 4. Create Beta Dossier
|
|
129
|
+
beta_dossier = {
|
|
130
|
+
"agent": "Agent Beta",
|
|
131
|
+
"scope_id": "scope_test",
|
|
132
|
+
"falsification_claims": [],
|
|
133
|
+
"methodological_critiques": [
|
|
134
|
+
{
|
|
135
|
+
"target_assertion": "Poseidon prover executes in 184ms",
|
|
136
|
+
"critique": "Benchmark excludes PCIe host bus latency"
|
|
137
|
+
}
|
|
138
|
+
],
|
|
139
|
+
"negative_knowledge": []
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
self.state_machine.record_agent_completion("scope_test", "alpha", alpha_dossier)
|
|
143
|
+
self.state_machine.record_agent_completion("scope_test", "beta", beta_dossier)
|
|
144
|
+
|
|
145
|
+
# 5. Run Auditor
|
|
146
|
+
report = self.auditor.audit_scope("scope_test")
|
|
147
|
+
summary = report["summary"]
|
|
148
|
+
|
|
149
|
+
self.assertEqual(summary["verified_passed"], 1)
|
|
150
|
+
self.assertEqual(summary["unverified_rejected"], 1)
|
|
151
|
+
self.assertEqual(summary["negative_knowledge_count"], 1)
|
|
152
|
+
self.assertGreater(summary["divergence_score"], 0.0)
|
|
153
|
+
|
|
154
|
+
# Check synthesis markdown generated
|
|
155
|
+
synth_file = self.test_dir / "scratchpads" / "scope_test" / "scope_synthesis.md"
|
|
156
|
+
self.assertTrue(synth_file.exists())
|
|
157
|
+
with open(synth_file, "r") as f:
|
|
158
|
+
synth_content = f.read()
|
|
159
|
+
self.assertIn("PURGED", synth_content)
|
|
160
|
+
self.assertIn("Poseidon prover executes in 184ms", synth_content)
|
|
161
|
+
|
|
162
|
+
def test_mock_swarm_runner_end_to_end(self):
|
|
163
|
+
runner = SwarmRunner(base_dir=self.test_dir, mock_mode=True)
|
|
164
|
+
runner.run_swarm("Evaluate hardware prover latency")
|
|
165
|
+
|
|
166
|
+
manifest = runner.state_machine.load_global_manifest()
|
|
167
|
+
self.assertEqual(manifest["status"], SessionStatus.COMPLETED.value)
|
|
168
|
+
|
|
169
|
+
final_report = self.test_dir / "final_synthesis.md"
|
|
170
|
+
self.assertTrue(final_report.exists())
|
|
171
|
+
with open(final_report, "r") as f:
|
|
172
|
+
report_text = f.read()
|
|
173
|
+
self.assertIn("Master Epistemic Research Report", report_text)
|
|
174
|
+
self.assertIn("Swarm Epistemic Audit Totals", report_text)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
if __name__ == "__main__":
|
|
178
|
+
unittest.main()
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: grilling
|
|
3
|
+
description: Socratic grilling and assumption-inversion skill for deep research. Uses Matt Pocock-style design trees to explore the problem frontier divergently before committing to search queries.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Socratic Grilling & Divergent Research Framing
|
|
7
|
+
|
|
8
|
+
Interview the user relentlessly until you reach an airtight, shared understanding of the research scope. Map the problem as a **design tree**: every foundational assumption branches into the technical decisions and empirical hypotheses that hang off it.
|
|
9
|
+
|
|
10
|
+
## 1. THE FRONTIER METHODOLOGY
|
|
11
|
+
|
|
12
|
+
1. Work the tree in **rounds**.
|
|
13
|
+
2. The **frontier** is every decision whose prerequisites are already settled: the questions you can ask *now* without guessing at answers you haven't heard yet.
|
|
14
|
+
3. Ask the whole frontier in one round: number each question and provide your recommended answer.
|
|
15
|
+
4. Then wait for the user's answers before moving to the next round.
|
|
16
|
+
|
|
17
|
+
## 2. FORMATTING A ROUND
|
|
18
|
+
|
|
19
|
+
```
|
|
20
|
+
❓ **Q1 - <Question Title>**: <Question context, premise inversion, trade-offs, multiple options>
|
|
21
|
+
|
|
22
|
+
➡️ **Recommended**: <Your recommended answer with rationale>
|
|
23
|
+
|
|
24
|
+
---
|
|
25
|
+
|
|
26
|
+
❓ **Q2 - <Question Title>**: <Question context, trade-offs>
|
|
27
|
+
|
|
28
|
+
➡️ **Recommended**: <Your recommended answer with rationale>
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
## 3. FACTUAL VS. DECISIONAL SEPARATION
|
|
32
|
+
|
|
33
|
+
- **Facts are the agent's job**: When a frontier question hinges on an empirical fact (e.g. library benchmarks, API specs, hardware limits), **DO NOT ASK THE USER**. Dispatch a tool call or subagent to look it up in the codebase or online.
|
|
34
|
+
- **Decisions are the user's**: High-level trade-offs, architectural philosophy, threat models, and priority ranking belong to the user. Put each decision to them clearly.
|
|
35
|
+
|
|
36
|
+
## 4. ASSUMPTION INVERSION TACTICS
|
|
37
|
+
|
|
38
|
+
Always challenge default premises in Round 1:
|
|
39
|
+
- *Inversion*: What if the primary objective is rendered obsolete by a radical alternative?
|
|
40
|
+
- *Scale Extremes*: What breaks at 100x scale? What breaks at 0 resources?
|
|
41
|
+
- *Adversarial Posture*: How would an intelligent adversary exploit or falsify this design?
|
|
42
|
+
|
|
43
|
+
## 5. FRONTIER RESOLUTION & FREEZING
|
|
44
|
+
|
|
45
|
+
When every branch of the design tree has been visited and the frontier is empty:
|
|
46
|
+
1. Summarize the settled constraints.
|
|
47
|
+
2. Save the settled state to `.research/frontier.json` using `python3 skills/grilling/socratic_tree.py --export`.
|
|
48
|
+
3. Hand off the settled frontier to the **Swarm Orchestrator** to begin empirical dialectic execution.
|
|
File without changes
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Socratic Tree & Decision Frontier Engine for Epistemic Swarm.
|
|
4
|
+
Implements Matt Pocock-style design tree traversal to isolate the active decision frontier.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import sys
|
|
8
|
+
import json
|
|
9
|
+
import argparse
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Dict, List, Optional, Any
|
|
12
|
+
|
|
13
|
+
class DecisionNode:
|
|
14
|
+
def __init__(self, node_id: str, title: str, question: str,
|
|
15
|
+
recommended: str, options: Optional[List[str]] = None,
|
|
16
|
+
prerequisites: Optional[List[str]] = None):
|
|
17
|
+
self.node_id = node_id
|
|
18
|
+
self.title = title
|
|
19
|
+
self.question = question
|
|
20
|
+
self.recommended = recommended
|
|
21
|
+
self.options = options or []
|
|
22
|
+
self.prerequisites = prerequisites or []
|
|
23
|
+
self.settled_answer: Optional[str] = None
|
|
24
|
+
|
|
25
|
+
def is_settled(self) -> bool:
|
|
26
|
+
return self.settled_answer is not None
|
|
27
|
+
|
|
28
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
29
|
+
return {
|
|
30
|
+
"node_id": self.node_id,
|
|
31
|
+
"title": self.title,
|
|
32
|
+
"question": self.question,
|
|
33
|
+
"recommended": self.recommended,
|
|
34
|
+
"options": self.options,
|
|
35
|
+
"prerequisites": self.prerequisites,
|
|
36
|
+
"settled_answer": self.settled_answer
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
@classmethod
|
|
40
|
+
def from_dict(cls, data: Dict[str, Any]) -> 'DecisionNode':
|
|
41
|
+
node = cls(
|
|
42
|
+
node_id=data["node_id"],
|
|
43
|
+
title=data["title"],
|
|
44
|
+
question=data["question"],
|
|
45
|
+
recommended=data["recommended"],
|
|
46
|
+
options=data.get("options", []),
|
|
47
|
+
prerequisites=data.get("prerequisites", [])
|
|
48
|
+
)
|
|
49
|
+
node.settled_answer = data.get("settled_answer")
|
|
50
|
+
return node
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class DesignTree:
|
|
54
|
+
def __init__(self, objective: str):
|
|
55
|
+
self.objective = objective
|
|
56
|
+
self.nodes: Dict[str, DecisionNode] = {}
|
|
57
|
+
|
|
58
|
+
def add_node(self, node: DecisionNode):
|
|
59
|
+
self.nodes[node.node_id] = node
|
|
60
|
+
|
|
61
|
+
def compute_frontier(self) -> List[DecisionNode]:
|
|
62
|
+
"""
|
|
63
|
+
The frontier is all unsettled nodes whose prerequisites are ALL settled.
|
|
64
|
+
"""
|
|
65
|
+
frontier = []
|
|
66
|
+
for node in self.nodes.values():
|
|
67
|
+
if node.is_settled():
|
|
68
|
+
continue
|
|
69
|
+
prereqs_met = True
|
|
70
|
+
for prereq_id in node.prerequisites:
|
|
71
|
+
prereq_node = self.nodes.get(prereq_id)
|
|
72
|
+
if not prereq_node or not prereq_node.is_settled():
|
|
73
|
+
prereqs_met = False
|
|
74
|
+
break
|
|
75
|
+
if prereqs_met:
|
|
76
|
+
frontier.append(node)
|
|
77
|
+
return frontier
|
|
78
|
+
|
|
79
|
+
def settle_node(self, node_id: str, answer: str):
|
|
80
|
+
if node_id in self.nodes:
|
|
81
|
+
self.nodes[node_id].settled_answer = answer
|
|
82
|
+
|
|
83
|
+
def is_complete(self) -> bool:
|
|
84
|
+
return len(self.compute_frontier()) == 0 and all(n.is_settled() for n in self.nodes.values())
|
|
85
|
+
|
|
86
|
+
def export_frontier_json(self, output_path: Path):
|
|
87
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
88
|
+
data = {
|
|
89
|
+
"objective": self.objective,
|
|
90
|
+
"is_complete": self.is_complete(),
|
|
91
|
+
"nodes": {k: v.to_dict() for k, v in self.nodes.items()},
|
|
92
|
+
"settled_constraints": {
|
|
93
|
+
k: v.settled_answer for k, v in self.nodes.items() if v.is_settled()
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
with open(output_path, "w", encoding="utf-8") as f:
|
|
97
|
+
json.dump(data, f, indent=2)
|
|
98
|
+
|
|
99
|
+
@classmethod
|
|
100
|
+
def load_from_json(cls, file_path: Path) -> 'DesignTree':
|
|
101
|
+
with open(file_path, "r", encoding="utf-8") as f:
|
|
102
|
+
data = json.load(f)
|
|
103
|
+
tree = cls(objective=data["objective"])
|
|
104
|
+
for k, v in data.get("nodes", {}).items():
|
|
105
|
+
tree.add_node(DecisionNode.from_dict(v))
|
|
106
|
+
return tree
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def main():
|
|
110
|
+
parser = argparse.ArgumentParser(description="Epistemic Swarm Socratic Decision Tree")
|
|
111
|
+
parser.add_argument("--objective", type=str, help="Research objective")
|
|
112
|
+
parser.add_argument("--file", type=str, default=".research/frontier.json", help="Path to frontier.json")
|
|
113
|
+
parser.add_argument("--show-frontier", action="store_true", help="Print the current decision frontier")
|
|
114
|
+
parser.add_argument("--settle", nargs=2, metavar=("NODE_ID", "ANSWER"), help="Settle a decision node")
|
|
115
|
+
|
|
116
|
+
args = parser.parse_args()
|
|
117
|
+
frontier_path = Path(args.file)
|
|
118
|
+
|
|
119
|
+
if frontier_path.exists():
|
|
120
|
+
tree = DesignTree.load_from_json(frontier_path)
|
|
121
|
+
else:
|
|
122
|
+
objective = args.objective or "Epistemic Swarm Research Objective"
|
|
123
|
+
tree = DesignTree(objective=objective)
|
|
124
|
+
|
|
125
|
+
if args.settle:
|
|
126
|
+
node_id, answer = args.settle
|
|
127
|
+
tree.settle_node(node_id, answer)
|
|
128
|
+
tree.export_frontier_json(frontier_path)
|
|
129
|
+
print(f"Settled {node_id} -> {answer}")
|
|
130
|
+
|
|
131
|
+
frontier = tree.compute_frontier()
|
|
132
|
+
if args.show_frontier or not args.settle:
|
|
133
|
+
print(f"\n🎯 Objective: {tree.objective}")
|
|
134
|
+
print(f"📊 Total Nodes: {len(tree.nodes)} | Settled: {sum(1 for n in tree.nodes.values() if n.is_settled())}")
|
|
135
|
+
if not frontier:
|
|
136
|
+
if tree.nodes and tree.is_complete():
|
|
137
|
+
print("✅ Frontier is EMPTY. All prerequisite branches are fully settled!")
|
|
138
|
+
else:
|
|
139
|
+
print("ℹ️ No active frontier nodes. Define new decision nodes to begin grilling.")
|
|
140
|
+
else:
|
|
141
|
+
print(f"\n⚡ Current Active Frontier ({len(frontier)} questions ready):")
|
|
142
|
+
for idx, node in enumerate(frontier, 1):
|
|
143
|
+
print(f"\n❓ Q{idx} [{node.node_id}] - {node.title}")
|
|
144
|
+
print(f" {node.question}")
|
|
145
|
+
print(f" ➡️ Recommended: {node.recommended}")
|
|
146
|
+
|
|
147
|
+
if __name__ == "__main__":
|
|
148
|
+
main()
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: research-cache
|
|
3
|
+
description: Content-addressed document caching and quote verification skill. Hashes retrieved web pages and academic papers to SHA-256 for mathematical auditability.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Content-Addressed Research Cache & Verification
|
|
7
|
+
|
|
8
|
+
To maintain epistemic integrity, every document fetched from the web, arXiv, or technical docs must be cached locally with a content-addressed SHA-256 fingerprint before its claims can be cited.
|
|
9
|
+
|
|
10
|
+
## 1. CACHING A SOURCE
|
|
11
|
+
When you fetch or scrape a URL:
|
|
12
|
+
```bash
|
|
13
|
+
python3 skills/research-cache/hasher.py cache \
|
|
14
|
+
--url "https://arxiv.org/abs/2407.21783" \
|
|
15
|
+
--title "Llama 3 Herd of Models" \
|
|
16
|
+
--content "$(cat fetched_paper.md)"
|
|
17
|
+
```
|
|
18
|
+
This prints the content hash:
|
|
19
|
+
```
|
|
20
|
+
[CACHED] 3f8a9e21... -> .research/sources/3f8a9e21....md
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
## 2. CITING WITH HASHES
|
|
24
|
+
In your dossiers and markdown reports, cite the claim using the hash:
|
|
25
|
+
`[VERIFIED: 3f8a9e21]`
|
|
26
|
+
|
|
27
|
+
Ensure that any `verbatim_quote` you provide is an exact substring from the cached markdown document.
|
|
28
|
+
|
|
29
|
+
## 3. AUDITING A QUOTE
|
|
30
|
+
The Epistemic Auditor verifies claims using:
|
|
31
|
+
```bash
|
|
32
|
+
python3 skills/research-cache/hasher.py verify \
|
|
33
|
+
--hash "3f8a9e21..." \
|
|
34
|
+
--quote "Our FPGA pipeline executes the Poseidon round constraints in 184ms"
|
|
35
|
+
```
|
|
36
|
+
If the quote does not match, the claim is rejected and flagged as unverified.
|
|
File without changes
|
|
Binary file
|
|
Binary file
|