@heretek-ai/epistemic-swarm 0.2.0 → 0.2.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +36 -9
- package/.claude-plugin/plugin.json +47 -41
- package/MARKETPLACE.md +93 -0
- package/README.md +61 -8
- package/bin/cli.js +58 -0
- package/config/mcp_launcher.py +215 -0
- package/config/opencode-snippet.json +39 -3
- package/extensions/pi/index.js +107 -4
- package/hooks/hooks.json +24 -0
- package/package.json +26 -3
- package/plugins/opencode/index.js +159 -5
- package/plugins/research-cache/.claude-plugin/plugin.json +15 -0
- package/plugins/socratic-grilling/.claude-plugin/plugin.json +15 -0
- package/plugins/socratic-grilling/skills/grilling/SKILL.md +48 -0
- package/plugins/socratic-grilling/skills/grilling/__init__.py +0 -0
- package/plugins/socratic-grilling/skills/grilling/socratic_tree.py +148 -0
- package/prompts/agent_code_auditor.md +88 -0
- package/prompts/agent_oss_scout.md +78 -0
- package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
- package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
- package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
- package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
- package/runner/research_swarm.py +230 -81
- package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
- package/runner/tests/test_swarm.py +88 -0
- package/skills/code_audit/SKILL.md +45 -0
- package/skills/epistemic_search/SKILL.md +35 -0
- package/skills/epistemic_search/__init__.py +1 -0
- package/skills/epistemic_search/scripts/fetch.py +157 -0
- package/skills/epistemic_search/scripts/search.py +162 -0
- package/skills/oss_scout/SKILL.md +49 -0
- package/skills/research_cache/SKILL.md +36 -0
- package/skills/research_cache/__init__.py +0 -0
- package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/{research-cache → research_cache}/__pycache__/hasher.cpython-311.pyc +0 -0
- package/skills/research_cache/hasher.py +195 -0
- package/skills/swarm_config/SKILL.md +72 -0
- package/skills/swarm_config/__init__.py +4 -0
- package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
- package/skills/swarm_config/configure.py +167 -0
- package/skills/research-cache/__pycache__/__init__.cpython-311.pyc +0 -0
- /package/{skills/research-cache → plugins/research-cache/skills/research_cache}/SKILL.md +0 -0
- /package/{skills/research-cache → plugins/research-cache/skills/research_cache}/__init__.py +0 -0
- /package/{skills/research-cache → plugins/research-cache/skills/research_cache}/hasher.py +0 -0
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Zero-API-Key Web Content Fetcher with Automatic Content-Addressed SHA-256 Caching.
|
|
4
|
+
Fetches web content, cleans HTML to readable Markdown, and persists directly into
|
|
5
|
+
.research/sources/<sha256>.md for epistemic auditability.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import sys
|
|
9
|
+
import os
|
|
10
|
+
import re
|
|
11
|
+
import json
|
|
12
|
+
import urllib.request
|
|
13
|
+
import urllib.parse
|
|
14
|
+
from html.parser import HTMLParser
|
|
15
|
+
|
|
16
|
+
# Add project root to sys.path to import SourceHasher
|
|
17
|
+
PKG_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..", ".."))
|
|
18
|
+
if PKG_ROOT not in sys.path:
|
|
19
|
+
sys.path.insert(0, PKG_ROOT)
|
|
20
|
+
|
|
21
|
+
try:
|
|
22
|
+
from skills.research_cache.hasher import SourceHasher
|
|
23
|
+
except ImportError:
|
|
24
|
+
SourceHasher = None
|
|
25
|
+
|
|
26
|
+
USER_AGENT = "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
|
27
|
+
|
|
28
|
+
class HTMLToMarkdownExtractor(HTMLParser):
|
|
29
|
+
def __init__(self):
|
|
30
|
+
super().__init__()
|
|
31
|
+
self.in_script_or_style = False
|
|
32
|
+
self.title = ""
|
|
33
|
+
self.in_title = False
|
|
34
|
+
self.text_chunks = []
|
|
35
|
+
self.headings = []
|
|
36
|
+
|
|
37
|
+
def handle_starttag(self, tag, attrs):
|
|
38
|
+
if tag in ("script", "style", "noscript", "svg", "header", "footer", "nav"):
|
|
39
|
+
self.in_script_or_style = True
|
|
40
|
+
elif tag == "title":
|
|
41
|
+
self.in_title = True
|
|
42
|
+
elif tag in ("h1", "h2", "h3", "h4", "h5", "h6"):
|
|
43
|
+
self.text_chunks.append(f"\n\n{'#' * int(tag[1])} ")
|
|
44
|
+
elif tag in ("p", "div", "article", "section"):
|
|
45
|
+
self.text_chunks.append("\n\n")
|
|
46
|
+
elif tag == "li":
|
|
47
|
+
self.text_chunks.append("\n- ")
|
|
48
|
+
elif tag == "br":
|
|
49
|
+
self.text_chunks.append("\n")
|
|
50
|
+
|
|
51
|
+
def handle_endtag(self, tag):
|
|
52
|
+
if tag in ("script", "style", "noscript", "svg", "header", "footer", "nav"):
|
|
53
|
+
self.in_script_or_style = False
|
|
54
|
+
elif tag == "title":
|
|
55
|
+
self.in_title = False
|
|
56
|
+
elif tag in ("p", "div", "article", "section"):
|
|
57
|
+
self.text_chunks.append("\n")
|
|
58
|
+
|
|
59
|
+
def handle_data(self, data):
|
|
60
|
+
if self.in_script_or_style:
|
|
61
|
+
return
|
|
62
|
+
if self.in_title:
|
|
63
|
+
self.title += data
|
|
64
|
+
else:
|
|
65
|
+
self.text_chunks.append(data)
|
|
66
|
+
|
|
67
|
+
def get_markdown(self) -> str:
|
|
68
|
+
raw_text = "".join(self.text_chunks)
|
|
69
|
+
# Clean multiple blank lines
|
|
70
|
+
clean_text = re.sub(r"\n{3,}", "\n\n", raw_text).strip()
|
|
71
|
+
return clean_text
|
|
72
|
+
|
|
73
|
+
def fetch_and_cache(url: str, research_dir: str = ".research") -> str:
|
|
74
|
+
headers = {
|
|
75
|
+
"User-Agent": USER_AGENT,
|
|
76
|
+
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
|
77
|
+
"Accept-Language": "en-US,en;q=0.5",
|
|
78
|
+
}
|
|
79
|
+
req = urllib.request.Request(url, headers=headers)
|
|
80
|
+
try:
|
|
81
|
+
with urllib.request.urlopen(req, timeout=20) as resp:
|
|
82
|
+
content_type = resp.headers.get("Content-Type", "")
|
|
83
|
+
raw_bytes = resp.read()
|
|
84
|
+
html = raw_bytes.decode("utf-8", errors="replace")
|
|
85
|
+
except Exception as e:
|
|
86
|
+
sys.stderr.write(f"[epistemic-fetch] Error fetching URL {url}: {e}\n")
|
|
87
|
+
return f"Error fetching {url}: {e}"
|
|
88
|
+
|
|
89
|
+
parser = HTMLToMarkdownExtractor()
|
|
90
|
+
parser.feed(html)
|
|
91
|
+
page_title = parser.title.strip() or url
|
|
92
|
+
body_markdown = parser.get_markdown()
|
|
93
|
+
|
|
94
|
+
# Cache into .research/sources/<sha256>.md
|
|
95
|
+
sha256_hash = "uncached"
|
|
96
|
+
cache_file = "not saved"
|
|
97
|
+
if SourceHasher:
|
|
98
|
+
from pathlib import Path
|
|
99
|
+
hasher = SourceHasher(base_dir=Path(research_dir))
|
|
100
|
+
sha256_hash = hasher.store_source(
|
|
101
|
+
url=url,
|
|
102
|
+
title=page_title,
|
|
103
|
+
content=body_markdown
|
|
104
|
+
)
|
|
105
|
+
cache_file = str(hasher.sources_dir / f"{sha256_hash}.md")
|
|
106
|
+
else:
|
|
107
|
+
import hashlib
|
|
108
|
+
sha256_hash = hashlib.sha256(body_markdown.encode("utf-8")).hexdigest()
|
|
109
|
+
|
|
110
|
+
output = []
|
|
111
|
+
output.append(f"<!-- EPISTEMIC_SOURCE_HASH: {sha256_hash} -->")
|
|
112
|
+
output.append(f"<!-- CACHED_AT: {cache_file} -->")
|
|
113
|
+
output.append(f"# {page_title}\n")
|
|
114
|
+
output.append(f"**Source URL:** {url}")
|
|
115
|
+
output.append(f"**Content SHA-256:** `{sha256_hash}`")
|
|
116
|
+
output.append(f"**Verification Tag:** `[VERIFIED: {sha256_hash[:16]}]`\n")
|
|
117
|
+
output.append(body_markdown)
|
|
118
|
+
|
|
119
|
+
return "\n".join(output)
|
|
120
|
+
|
|
121
|
+
def main():
|
|
122
|
+
url = ""
|
|
123
|
+
research_dir = ".research"
|
|
124
|
+
|
|
125
|
+
if not sys.stdin.isatty():
|
|
126
|
+
try:
|
|
127
|
+
stdin_data = sys.stdin.read().strip()
|
|
128
|
+
if stdin_data:
|
|
129
|
+
try:
|
|
130
|
+
payload = json.loads(stdin_data)
|
|
131
|
+
url = payload.get("url", "")
|
|
132
|
+
research_dir = payload.get("research_dir", research_dir)
|
|
133
|
+
except json.JSONDecodeError:
|
|
134
|
+
url = stdin_data
|
|
135
|
+
except Exception:
|
|
136
|
+
pass
|
|
137
|
+
|
|
138
|
+
args = sys.argv[1:]
|
|
139
|
+
idx = 0
|
|
140
|
+
while idx < len(args):
|
|
141
|
+
arg = args[idx]
|
|
142
|
+
if arg == "--dir" and idx + 1 < len(args):
|
|
143
|
+
research_dir = args[idx + 1]
|
|
144
|
+
idx += 1
|
|
145
|
+
elif not url and not arg.startswith("--"):
|
|
146
|
+
url = arg
|
|
147
|
+
idx += 1
|
|
148
|
+
|
|
149
|
+
if not url:
|
|
150
|
+
sys.stderr.write("Usage: fetch.py [options] <url>\nOr pipe JSON: echo '{\"url\":\"https://...\"}' | fetch.py\n")
|
|
151
|
+
sys.exit(1)
|
|
152
|
+
|
|
153
|
+
result = fetch_and_cache(url=url, research_dir=research_dir)
|
|
154
|
+
print(result)
|
|
155
|
+
|
|
156
|
+
if __name__ == "__main__":
|
|
157
|
+
main()
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Zero-API-Key Web Search Engine for IUMBTEMS.
|
|
4
|
+
DuckDuckGo HTML/lite parser with domain filtering and Claude Code XML formatting.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import sys
|
|
8
|
+
import json
|
|
9
|
+
import re
|
|
10
|
+
import urllib.request
|
|
11
|
+
import urllib.parse
|
|
12
|
+
from typing import List, Dict, Optional
|
|
13
|
+
|
|
14
|
+
USER_AGENT = "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
|
15
|
+
|
|
16
|
+
def search_duckduckgo(
|
|
17
|
+
query: str,
|
|
18
|
+
allowed_domains: Optional[List[str]] = None,
|
|
19
|
+
blocked_domains: Optional[List[str]] = None,
|
|
20
|
+
max_results: int = 10
|
|
21
|
+
) -> List[Dict[str, str]]:
|
|
22
|
+
"""Execute search query against DuckDuckGo Lite without API keys."""
|
|
23
|
+
effective_query = query.strip()
|
|
24
|
+
if allowed_domains:
|
|
25
|
+
domain_filters = " " + " OR ".join(f"site:{d}" for d in allowed_domains)
|
|
26
|
+
effective_query += domain_filters
|
|
27
|
+
if blocked_domains:
|
|
28
|
+
domain_filters = " " + " ".join(f"-site:{d}" for d in blocked_domains)
|
|
29
|
+
effective_query += domain_filters
|
|
30
|
+
|
|
31
|
+
url = "https://lite.duckduckgo.com/lite/"
|
|
32
|
+
data = urllib.parse.urlencode({"q": effective_query}).encode("utf-8")
|
|
33
|
+
headers = {
|
|
34
|
+
"User-Agent": USER_AGENT,
|
|
35
|
+
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
|
36
|
+
"Accept-Language": "en-US,en;q=0.5",
|
|
37
|
+
"Referer": "https://lite.duckduckgo.com/",
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
req = urllib.request.Request(url, data=data, headers=headers)
|
|
41
|
+
try:
|
|
42
|
+
with urllib.request.urlopen(req, timeout=15) as resp:
|
|
43
|
+
html = resp.read().decode("utf-8", errors="replace")
|
|
44
|
+
except Exception as e:
|
|
45
|
+
sys.stderr.write(f"[epistemic-search] Network error fetching search results: {e}\n")
|
|
46
|
+
return []
|
|
47
|
+
|
|
48
|
+
# Parse result links and snippets from DuckDuckGo Lite table rows
|
|
49
|
+
links = re.findall(
|
|
50
|
+
r"<a[^>]*href=['\"]([^'\"]+)['\"][^>]*class=['\"]result-link['\"][^>]*>(.*?)</a>",
|
|
51
|
+
html,
|
|
52
|
+
re.DOTALL
|
|
53
|
+
)
|
|
54
|
+
snippets = re.findall(
|
|
55
|
+
r"<td[^>]+class=['\"]result-snippet['\"][^>]*>(.*?)</td>",
|
|
56
|
+
html,
|
|
57
|
+
re.DOTALL
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
results = []
|
|
61
|
+
for (href, title_html), snip_html in zip(links, snippets):
|
|
62
|
+
# Unwrap DuckDuckGo redirect if present
|
|
63
|
+
m = re.search(r"uddg=([^&]+)", href)
|
|
64
|
+
if m:
|
|
65
|
+
clean_url = urllib.parse.unquote(m.group(1))
|
|
66
|
+
else:
|
|
67
|
+
clean_url = href
|
|
68
|
+
|
|
69
|
+
if not clean_url.startswith("http"):
|
|
70
|
+
continue
|
|
71
|
+
|
|
72
|
+
clean_title = re.sub(r"<[^>]+>", "", title_html).strip()
|
|
73
|
+
clean_snippet = re.sub(r"<[^>]+>", "", snip_html).strip()
|
|
74
|
+
|
|
75
|
+
# Check domain filtering manually in case DDG syntax didn't catch all
|
|
76
|
+
parsed_url = urllib.parse.urlparse(clean_url)
|
|
77
|
+
hostname = parsed_url.hostname or ""
|
|
78
|
+
if allowed_domains and not any(hostname == d or hostname.endswith("." + d) for d in allowed_domains):
|
|
79
|
+
continue
|
|
80
|
+
if blocked_domains and any(hostname == d or hostname.endswith("." + d) for d in blocked_domains):
|
|
81
|
+
continue
|
|
82
|
+
|
|
83
|
+
results.append({
|
|
84
|
+
"title": clean_title,
|
|
85
|
+
"url": clean_url,
|
|
86
|
+
"snippet": clean_snippet
|
|
87
|
+
})
|
|
88
|
+
|
|
89
|
+
if len(results) >= max_results:
|
|
90
|
+
break
|
|
91
|
+
|
|
92
|
+
return results
|
|
93
|
+
|
|
94
|
+
def format_xml(results: List[Dict[str, str]]) -> str:
|
|
95
|
+
"""Format results as Claude Code <search_results> XML."""
|
|
96
|
+
lines = ["<search_results>"]
|
|
97
|
+
for r in results:
|
|
98
|
+
lines.append(" <result>")
|
|
99
|
+
lines.append(f" <title>{urllib.parse.quote(r['title'])}</title>")
|
|
100
|
+
lines.append(f" <url>{r['url']}</url>")
|
|
101
|
+
lines.append(f" <snippet>{r['snippet']}</snippet>")
|
|
102
|
+
lines.append(" </result>")
|
|
103
|
+
lines.append("</search_results>")
|
|
104
|
+
return "\n".join(lines)
|
|
105
|
+
|
|
106
|
+
def main():
|
|
107
|
+
query = ""
|
|
108
|
+
allowed_domains = None
|
|
109
|
+
blocked_domains = None
|
|
110
|
+
output_json = False
|
|
111
|
+
|
|
112
|
+
# Check stdin first
|
|
113
|
+
if not sys.stdin.isatty():
|
|
114
|
+
try:
|
|
115
|
+
stdin_data = sys.stdin.read().strip()
|
|
116
|
+
if stdin_data:
|
|
117
|
+
try:
|
|
118
|
+
payload = json.loads(stdin_data)
|
|
119
|
+
query = payload.get("query", "")
|
|
120
|
+
allowed_domains = payload.get("allowed_domains")
|
|
121
|
+
blocked_domains = payload.get("blocked_domains")
|
|
122
|
+
except json.JSONDecodeError:
|
|
123
|
+
query = stdin_data
|
|
124
|
+
except Exception:
|
|
125
|
+
pass
|
|
126
|
+
|
|
127
|
+
# Command line args override
|
|
128
|
+
args = sys.argv[1:]
|
|
129
|
+
idx = 0
|
|
130
|
+
while idx < len(args):
|
|
131
|
+
arg = args[idx]
|
|
132
|
+
if arg == "--json":
|
|
133
|
+
output_json = True
|
|
134
|
+
elif arg == "--allowed-domain" and idx + 1 < len(args):
|
|
135
|
+
allowed_domains = allowed_domains or []
|
|
136
|
+
allowed_domains.append(args[idx + 1])
|
|
137
|
+
idx += 1
|
|
138
|
+
elif arg == "--blocked-domain" and idx + 1 < len(args):
|
|
139
|
+
blocked_domains = blocked_domains or []
|
|
140
|
+
blocked_domains.append(args[idx + 1])
|
|
141
|
+
idx += 1
|
|
142
|
+
elif not query and not arg.startswith("--"):
|
|
143
|
+
query = arg
|
|
144
|
+
idx += 1
|
|
145
|
+
|
|
146
|
+
if not query:
|
|
147
|
+
sys.stderr.write("Usage: search.py [options] <query>\nOr pipe JSON: echo '{\"query\":\"...\"}' | search.py\n")
|
|
148
|
+
sys.exit(1)
|
|
149
|
+
|
|
150
|
+
results = search_duckduckgo(
|
|
151
|
+
query=query,
|
|
152
|
+
allowed_domains=allowed_domains,
|
|
153
|
+
blocked_domains=blocked_domains
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
if output_json:
|
|
157
|
+
print(json.dumps(results, indent=2))
|
|
158
|
+
else:
|
|
159
|
+
print(format_xml(results))
|
|
160
|
+
|
|
161
|
+
if __name__ == "__main__":
|
|
162
|
+
main()
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: oss-scout
|
|
3
|
+
description: Open-source software discovery, dependency vetting, and clean-room implementation scouting. Evaluates GitHub repositories, package ecosystems, licenses, and architecture to discover code to adopt or borrow.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Open Source Scout & Clean-Room Harvesting Engine
|
|
7
|
+
|
|
8
|
+
Scout the open-source software ecosystem to discover mature libraries, reference implementations, and algorithms for research-based development.
|
|
9
|
+
The scout deploys a **Discovery Scout (Thesis)** to find high-performance implementations and a **Licensing & Bloat Red-Teamer (Antithesis)** to protect your project against viral copyleft, unmaintained abandonware, and security CVEs.
|
|
10
|
+
|
|
11
|
+
## 1. Invoking an Open Source Scout
|
|
12
|
+
|
|
13
|
+
Search for open-source solutions to implement a feature:
|
|
14
|
+
```bash
|
|
15
|
+
iumbtems scout "Find high-throughput zero-dependency Raft consensus implementations in Rust or Go"
|
|
16
|
+
```
|
|
17
|
+
Or via npx:
|
|
18
|
+
```bash
|
|
19
|
+
npx @heretek-ai/epistemic-swarm scout "Scout vector database indexing algorithms with MIT or Apache-2.0 license"
|
|
20
|
+
```
|
|
21
|
+
Or directly with the python runner:
|
|
22
|
+
```bash
|
|
23
|
+
python3 runner/research_swarm.py --mode scout --objective "Explore clean-room alternatives to AGPL licensed search engines"
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
## 2. Dialectic Evaluation Workflow
|
|
27
|
+
|
|
28
|
+
1. **Discovery (Alpha)**
|
|
29
|
+
- Explores GitHub, GitLab, crates.io, PyPI, npm, and Go packages.
|
|
30
|
+
- Compares star velocity, release frequency, benchmark throughput, and API design.
|
|
31
|
+
- Automatically caches repository documentation into `.research/sources/<sha256>.md`.
|
|
32
|
+
|
|
33
|
+
2. **Adversarial Red-Teaming (Beta)**
|
|
34
|
+
- **License Contamination**: Flags AGPL/GPL requirements that could compromise proprietary or permissive codebases.
|
|
35
|
+
- **Maintenance Health**: Detects abandonware, stagnant commit logs, unresponsive PR queues, and solo maintainer risks.
|
|
36
|
+
- **Dependency Weight**: Audits transitive dependency explosion and bundle size.
|
|
37
|
+
- **Security Surface**: Scans for known CVEs and malicious package takeover vulnerabilities.
|
|
38
|
+
|
|
39
|
+
3. **Synthesis & Clean-Room Blueprints**
|
|
40
|
+
- Synthesizes findings into `.research/oss_scout_report.md` and `.research/oss_scout_dossier.json`.
|
|
41
|
+
- Produces a **Clean-Room Blueprint**: an algorithmic breakdown allowing in-tree implementation of the core feature without licensing entanglement.
|
|
42
|
+
|
|
43
|
+
## 3. Supported Package Ecosystems
|
|
44
|
+
- GitHub & GitLab Repositories
|
|
45
|
+
- Rust (crates.io)
|
|
46
|
+
- Python (PyPI)
|
|
47
|
+
- TypeScript / Node.js (npm)
|
|
48
|
+
- Go Modules
|
|
49
|
+
- C / C++ header-only libraries
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: research-cache
|
|
3
|
+
description: Content-addressed document caching and quote verification skill. Hashes retrieved web pages and academic papers to SHA-256 for mathematical auditability.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Content-Addressed Research Cache & Verification
|
|
7
|
+
|
|
8
|
+
To maintain epistemic integrity, every document fetched from the web, arXiv, or technical docs must be cached locally with a content-addressed SHA-256 fingerprint before its claims can be cited.
|
|
9
|
+
|
|
10
|
+
## 1. CACHING A SOURCE
|
|
11
|
+
When you fetch or scrape a URL:
|
|
12
|
+
```bash
|
|
13
|
+
python3 skills/research-cache/hasher.py cache \
|
|
14
|
+
--url "https://arxiv.org/abs/2407.21783" \
|
|
15
|
+
--title "Llama 3 Herd of Models" \
|
|
16
|
+
--content "$(cat fetched_paper.md)"
|
|
17
|
+
```
|
|
18
|
+
This prints the content hash:
|
|
19
|
+
```
|
|
20
|
+
[CACHED] 3f8a9e21... -> .research/sources/3f8a9e21....md
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
## 2. CITING WITH HASHES
|
|
24
|
+
In your dossiers and markdown reports, cite the claim using the hash:
|
|
25
|
+
`[VERIFIED: 3f8a9e21]`
|
|
26
|
+
|
|
27
|
+
Ensure that any `verbatim_quote` you provide is an exact substring from the cached markdown document.
|
|
28
|
+
|
|
29
|
+
## 3. AUDITING A QUOTE
|
|
30
|
+
The Epistemic Auditor verifies claims using:
|
|
31
|
+
```bash
|
|
32
|
+
python3 skills/research-cache/hasher.py verify \
|
|
33
|
+
--hash "3f8a9e21..." \
|
|
34
|
+
--quote "Our FPGA pipeline executes the Poseidon round constraints in 184ms"
|
|
35
|
+
```
|
|
36
|
+
If the quote does not match, the claim is rejected and flagged as unverified.
|
|
File without changes
|
|
Binary file
|
|
Binary file
|
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Content-Addressed Research Cache & Verbatim Quote Verification Engine.
|
|
4
|
+
Handles SHA-256 document hashing, metadata indexing, and substring auditing.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import os
|
|
8
|
+
import sys
|
|
9
|
+
import json
|
|
10
|
+
import hashlib
|
|
11
|
+
import re
|
|
12
|
+
import argparse
|
|
13
|
+
from datetime import datetime, timezone
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Dict, Any, Optional, Tuple
|
|
16
|
+
|
|
17
|
+
class SourceHasher:
|
|
18
|
+
def __init__(self, base_dir: Optional[Path] = None):
|
|
19
|
+
self.base_dir = base_dir or Path(".research")
|
|
20
|
+
self.sources_dir = self.base_dir / "sources"
|
|
21
|
+
self.sources_dir.mkdir(parents=True, exist_ok=True)
|
|
22
|
+
|
|
23
|
+
@staticmethod
|
|
24
|
+
def compute_sha256(content: str) -> str:
|
|
25
|
+
"""Compute standard hex SHA-256 hash of normalized UTF-8 string."""
|
|
26
|
+
normalized = content.strip().encode("utf-8")
|
|
27
|
+
return hashlib.sha256(normalized).hexdigest()
|
|
28
|
+
|
|
29
|
+
def store_source(self, url: str, content: str, title: Optional[str] = None,
|
|
30
|
+
tier: str = "WEB_DOCUMENT", metadata: Optional[Dict[str, Any]] = None) -> str:
|
|
31
|
+
"""Store content and metadata content-addressed by SHA-256."""
|
|
32
|
+
content_hash = self.compute_sha256(content)
|
|
33
|
+
md_path = self.sources_dir / f"{content_hash}.md"
|
|
34
|
+
json_path = self.sources_dir / f"{content_hash}.json"
|
|
35
|
+
|
|
36
|
+
# Write clean markdown
|
|
37
|
+
with open(md_path, "w", encoding="utf-8") as f:
|
|
38
|
+
f.write(content)
|
|
39
|
+
|
|
40
|
+
# Write metadata
|
|
41
|
+
meta = {
|
|
42
|
+
"hash": content_hash,
|
|
43
|
+
"url": url,
|
|
44
|
+
"title": title or "Untitled Source",
|
|
45
|
+
"tier": tier,
|
|
46
|
+
"cached_at": datetime.now(timezone.utc).isoformat(),
|
|
47
|
+
"byte_size": len(content.encode("utf-8")),
|
|
48
|
+
"char_count": len(content),
|
|
49
|
+
"custom_metadata": metadata or {}
|
|
50
|
+
}
|
|
51
|
+
with open(json_path, "w", encoding="utf-8") as f:
|
|
52
|
+
json.dump(meta, f, indent=2)
|
|
53
|
+
|
|
54
|
+
return content_hash
|
|
55
|
+
|
|
56
|
+
def get_source_content(self, content_hash: str) -> Optional[str]:
|
|
57
|
+
"""Retrieve stored markdown content by full or prefix hash."""
|
|
58
|
+
target_file = self._resolve_hash_file(content_hash, ".md")
|
|
59
|
+
if target_file and target_file.exists():
|
|
60
|
+
with open(target_file, "r", encoding="utf-8") as f:
|
|
61
|
+
return f.read()
|
|
62
|
+
return None
|
|
63
|
+
|
|
64
|
+
def get_source_metadata(self, content_hash: str) -> Optional[Dict[str, Any]]:
|
|
65
|
+
"""Retrieve stored metadata by full or prefix hash."""
|
|
66
|
+
target_file = self._resolve_hash_file(content_hash, ".json")
|
|
67
|
+
if target_file and target_file.exists():
|
|
68
|
+
with open(target_file, "r", encoding="utf-8") as f:
|
|
69
|
+
return json.load(f)
|
|
70
|
+
return None
|
|
71
|
+
|
|
72
|
+
def _resolve_hash_file(self, hash_prefix: str, extension: str) -> Optional[Path]:
|
|
73
|
+
"""Resolves exact or prefix hash to disk path."""
|
|
74
|
+
exact_path = self.sources_dir / f"{hash_prefix}{extension}"
|
|
75
|
+
if exact_path.exists():
|
|
76
|
+
return exact_path
|
|
77
|
+
|
|
78
|
+
# If prefix match
|
|
79
|
+
matches = list(self.sources_dir.glob(f"{hash_prefix}*{extension}"))
|
|
80
|
+
if len(matches) == 1:
|
|
81
|
+
return matches[0]
|
|
82
|
+
return None
|
|
83
|
+
|
|
84
|
+
@staticmethod
|
|
85
|
+
def normalize_text_for_search(text: str) -> str:
|
|
86
|
+
"""Collapse whitespace and normalize typography for substring matching."""
|
|
87
|
+
# Replace smart quotes and dashes
|
|
88
|
+
text = text.replace("“", '"').replace("”", '"').replace("‘", "'").replace("’", "'")
|
|
89
|
+
text = text.replace("—", "-").replace("–", "-")
|
|
90
|
+
# Collapse all whitespace to single spaces
|
|
91
|
+
return re.sub(r"\s+", " ", text).strip().lower()
|
|
92
|
+
|
|
93
|
+
def verify_quote(self, content_hash: str, quote: str) -> Tuple[bool, float, Optional[str]]:
|
|
94
|
+
"""
|
|
95
|
+
Verifies whether quote exists in cached document.
|
|
96
|
+
Returns: (is_verified, confidence_score, context_match)
|
|
97
|
+
"""
|
|
98
|
+
source_text = self.get_source_content(content_hash)
|
|
99
|
+
if not source_text:
|
|
100
|
+
return False, 0.0, f"Source hash '{content_hash}' not found in cache."
|
|
101
|
+
|
|
102
|
+
# 1. Exact match test
|
|
103
|
+
if quote.strip() in source_text:
|
|
104
|
+
return True, 1.0, "Exact substring match found."
|
|
105
|
+
|
|
106
|
+
# 2. Normalized whitespace match test
|
|
107
|
+
norm_source = self.normalize_text_for_search(source_text)
|
|
108
|
+
norm_quote = self.normalize_text_for_search(quote)
|
|
109
|
+
if norm_quote in norm_source:
|
|
110
|
+
return True, 0.98, "Normalized whitespace match found."
|
|
111
|
+
|
|
112
|
+
# 3. Sliding window token overlap test
|
|
113
|
+
quote_words = norm_quote.split()
|
|
114
|
+
if len(quote_words) < 4:
|
|
115
|
+
return False, 0.0, "Quote too short and not found verbatim."
|
|
116
|
+
|
|
117
|
+
window_size = len(quote_words)
|
|
118
|
+
source_words = norm_source.split()
|
|
119
|
+
best_score = 0.0
|
|
120
|
+
|
|
121
|
+
for i in range(max(1, len(source_words) - window_size + 1)):
|
|
122
|
+
window = source_words[i:i + window_size]
|
|
123
|
+
matches = sum(1 for w1, w2 in zip(quote_words, window) if w1 == w2)
|
|
124
|
+
score = matches / window_size
|
|
125
|
+
if score > best_score:
|
|
126
|
+
best_score = score
|
|
127
|
+
if best_score >= 0.90:
|
|
128
|
+
break
|
|
129
|
+
|
|
130
|
+
if best_score >= 0.88:
|
|
131
|
+
return True, best_score, f"High-confidence fuzzy match ({best_score:.2f})."
|
|
132
|
+
|
|
133
|
+
return False, best_score, f"Verification failed. Highest word overlap: {best_score:.2f}."
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def main():
|
|
137
|
+
parser = argparse.ArgumentParser(description="Epistemic Swarm Content Hasher & Quote Verifier")
|
|
138
|
+
subparsers = parser.add_subparsers(dest="command")
|
|
139
|
+
|
|
140
|
+
# Cache command
|
|
141
|
+
cache_parser = subparsers.add_parser("cache", help="Cache a document")
|
|
142
|
+
cache_parser.add_argument("--url", required=True, help="Original URL")
|
|
143
|
+
cache_parser.add_argument("--title", default="Untitled", help="Document Title")
|
|
144
|
+
cache_parser.add_argument("--tier", default="WEB_DOCUMENT", help="Source tier")
|
|
145
|
+
cache_parser.add_argument("--content", help="Raw text content (or read from stdin)")
|
|
146
|
+
cache_parser.add_argument("--dir", default=".research", help="Base .research directory")
|
|
147
|
+
|
|
148
|
+
# Verify command
|
|
149
|
+
verify_parser = subparsers.add_parser("verify", help="Verify a verbatim quote")
|
|
150
|
+
verify_parser.add_argument("--hash", required=True, help="Document SHA-256 hash")
|
|
151
|
+
verify_parser.add_argument("--quote", required=True, help="Verbatim quote to check")
|
|
152
|
+
verify_parser.add_argument("--dir", default=".research", help="Base .research directory")
|
|
153
|
+
|
|
154
|
+
# List command
|
|
155
|
+
list_parser = subparsers.add_parser("list", help="List cached sources")
|
|
156
|
+
list_parser.add_argument("--dir", default=".research", help="Base .research directory")
|
|
157
|
+
|
|
158
|
+
args = parser.parse_args()
|
|
159
|
+
hasher = SourceHasher(base_dir=Path(args.dir if hasattr(args, "dir") else ".research"))
|
|
160
|
+
|
|
161
|
+
if args.command == "cache":
|
|
162
|
+
content = args.content
|
|
163
|
+
if not content:
|
|
164
|
+
if not sys.stdin.isatty():
|
|
165
|
+
content = sys.stdin.read()
|
|
166
|
+
else:
|
|
167
|
+
print("Error: No content provided via --content or stdin.", file=sys.stderr)
|
|
168
|
+
sys.exit(1)
|
|
169
|
+
h = hasher.store_source(url=args.url, content=content, title=args.title, tier=args.tier)
|
|
170
|
+
print(f"[CACHED] {h} -> {args.title} ({args.url})")
|
|
171
|
+
|
|
172
|
+
elif args.command == "verify":
|
|
173
|
+
verified, conf, msg = hasher.verify_quote(content_hash=args.hash, quote=args.quote)
|
|
174
|
+
result = {
|
|
175
|
+
"hash": args.hash,
|
|
176
|
+
"verified": verified,
|
|
177
|
+
"confidence": conf,
|
|
178
|
+
"message": msg
|
|
179
|
+
}
|
|
180
|
+
print(json.dumps(result, indent=2))
|
|
181
|
+
sys.exit(0 if verified else 1)
|
|
182
|
+
|
|
183
|
+
elif args.command == "list":
|
|
184
|
+
sources = list(hasher.sources_dir.glob("*.json"))
|
|
185
|
+
print(f"Total Cached Sources: {len(sources)}")
|
|
186
|
+
for p in sources:
|
|
187
|
+
with open(p, "r", encoding="utf-8") as f:
|
|
188
|
+
d = json.load(f)
|
|
189
|
+
print(f"- [{d['hash'][:10]}...] {d['title']} ({d['url']})")
|
|
190
|
+
else:
|
|
191
|
+
parser.print_help()
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
if __name__ == "__main__":
|
|
195
|
+
main()
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: swarm-config
|
|
3
|
+
description: Dynamic configuration and settings skill for the Epistemic Swarm research harness. Manage search engines, research depth, dialectic iterations, operating modes (research, audit, scout), and license filters.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Epistemic Swarm Configuration & Parameter Tuning
|
|
7
|
+
|
|
8
|
+
Use this skill to inspect, tune, and persist research parameters into `.research/config.json`.
|
|
9
|
+
Settings are automatically loaded by the Swarm Runner, Epistemic Auditor, and CLI.
|
|
10
|
+
|
|
11
|
+
## 1. Interactive Configuration
|
|
12
|
+
|
|
13
|
+
To launch the interactive configuration prompt:
|
|
14
|
+
```bash
|
|
15
|
+
python3 skills/swarm_config/configure.py --interactive
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
This guides you through selecting:
|
|
19
|
+
1. **Primary Search Engine**:
|
|
20
|
+
- `duckduckgo` (Default, zero API key required, completely private)
|
|
21
|
+
- `brave` (Brave Search API for high-precision SERP results)
|
|
22
|
+
- `firecrawl` (Deep web scraping & JavaScript rendering)
|
|
23
|
+
- `searxng` (Self-hosted privacy metasearch aggregator)
|
|
24
|
+
2. **Research Depth & Dialectic Iterations**:
|
|
25
|
+
- `1` (Rapid brief, minimal token usage)
|
|
26
|
+
- `2` (Standard thesis vs. antithesis dialectic - recommended)
|
|
27
|
+
- `3` (Deep multi-pass verification)
|
|
28
|
+
- `4` (Exhaustive multi-scope investigation)
|
|
29
|
+
3. **Operating Mode**:
|
|
30
|
+
- `research` (Empirical literature & web synthesis)
|
|
31
|
+
- `audit` (Deep codebase architecture, security, and vulnerability red-teaming)
|
|
32
|
+
- `scout` (Open-source software discovery & clean-room harvesting)
|
|
33
|
+
- `hybrid` (Combined codebase audit + web research)
|
|
34
|
+
|
|
35
|
+
## 2. Direct CLI Configuration
|
|
36
|
+
|
|
37
|
+
Inspect the current active configuration:
|
|
38
|
+
```bash
|
|
39
|
+
python3 skills/swarm_config/configure.py --show
|
|
40
|
+
```
|
|
41
|
+
or via CLI:
|
|
42
|
+
```bash
|
|
43
|
+
iumbtems config
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Update parameters directly via flags:
|
|
47
|
+
```bash
|
|
48
|
+
# Set search engine to duckduckgo and depth to 3
|
|
49
|
+
python3 skills/swarm_config/configure.py --engine duckduckgo --depth 3
|
|
50
|
+
|
|
51
|
+
# Set operating mode to codebase audit
|
|
52
|
+
python3 skills/swarm_config/configure.py --mode audit
|
|
53
|
+
|
|
54
|
+
# Set divergence threshold
|
|
55
|
+
python3 skills/swarm_config/configure.py --divergence 0.8
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
## 3. Configuration Schema (`.research/config.json`)
|
|
59
|
+
|
|
60
|
+
The active project configuration is persisted at `.research/config.json`:
|
|
61
|
+
```json
|
|
62
|
+
{
|
|
63
|
+
"search_engine": "duckduckgo",
|
|
64
|
+
"max_iterations": 2,
|
|
65
|
+
"divergence_threshold": 0.75,
|
|
66
|
+
"mode": "research",
|
|
67
|
+
"cache_raw_markdown": true,
|
|
68
|
+
"license_whitelist": ["MIT", "Apache-2.0", "BSD-3-Clause", "ISC"],
|
|
69
|
+
"output_dir": ".research"
|
|
70
|
+
}
|
|
71
|
+
```
|
|
72
|
+
All swarm agents read this file at initialization time.
|