@heretek-ai/epistemic-swarm 0.2.0 → 0.2.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/.claude-plugin/marketplace.json +36 -9
  2. package/.claude-plugin/plugin.json +47 -41
  3. package/MARKETPLACE.md +93 -0
  4. package/README.md +61 -8
  5. package/bin/cli.js +58 -0
  6. package/config/mcp_launcher.py +215 -0
  7. package/config/opencode-snippet.json +39 -3
  8. package/extensions/pi/index.js +107 -4
  9. package/hooks/hooks.json +24 -0
  10. package/package.json +26 -3
  11. package/plugins/opencode/index.js +159 -5
  12. package/plugins/research-cache/.claude-plugin/plugin.json +15 -0
  13. package/plugins/socratic-grilling/.claude-plugin/plugin.json +15 -0
  14. package/plugins/socratic-grilling/skills/grilling/SKILL.md +48 -0
  15. package/plugins/socratic-grilling/skills/grilling/__init__.py +0 -0
  16. package/plugins/socratic-grilling/skills/grilling/socratic_tree.py +148 -0
  17. package/prompts/agent_code_auditor.md +88 -0
  18. package/prompts/agent_oss_scout.md +78 -0
  19. package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
  20. package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
  21. package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
  22. package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
  23. package/runner/research_swarm.py +230 -81
  24. package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
  25. package/runner/tests/test_swarm.py +88 -0
  26. package/skills/code_audit/SKILL.md +45 -0
  27. package/skills/epistemic_search/SKILL.md +35 -0
  28. package/skills/epistemic_search/__init__.py +1 -0
  29. package/skills/epistemic_search/scripts/fetch.py +157 -0
  30. package/skills/epistemic_search/scripts/search.py +162 -0
  31. package/skills/oss_scout/SKILL.md +49 -0
  32. package/skills/research_cache/SKILL.md +36 -0
  33. package/skills/research_cache/__init__.py +0 -0
  34. package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
  35. package/skills/{research-cache → research_cache}/__pycache__/hasher.cpython-311.pyc +0 -0
  36. package/skills/research_cache/hasher.py +195 -0
  37. package/skills/swarm_config/SKILL.md +72 -0
  38. package/skills/swarm_config/__init__.py +4 -0
  39. package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
  40. package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
  41. package/skills/swarm_config/configure.py +167 -0
  42. package/skills/research-cache/__pycache__/__init__.cpython-311.pyc +0 -0
  43. /package/{skills/research-cache → plugins/research-cache/skills/research_cache}/SKILL.md +0 -0
  44. /package/{skills/research-cache → plugins/research-cache/skills/research_cache}/__init__.py +0 -0
  45. /package/{skills/research-cache → plugins/research-cache/skills/research_cache}/hasher.py +0 -0
@@ -0,0 +1,157 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ Zero-API-Key Web Content Fetcher with Automatic Content-Addressed SHA-256 Caching.
4
+ Fetches web content, cleans HTML to readable Markdown, and persists directly into
5
+ .research/sources/<sha256>.md for epistemic auditability.
6
+ """
7
+
8
+ import sys
9
+ import os
10
+ import re
11
+ import json
12
+ import urllib.request
13
+ import urllib.parse
14
+ from html.parser import HTMLParser
15
+
16
+ # Add project root to sys.path to import SourceHasher
17
+ PKG_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..", ".."))
18
+ if PKG_ROOT not in sys.path:
19
+ sys.path.insert(0, PKG_ROOT)
20
+
21
+ try:
22
+ from skills.research_cache.hasher import SourceHasher
23
+ except ImportError:
24
+ SourceHasher = None
25
+
26
+ USER_AGENT = "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
27
+
28
+ class HTMLToMarkdownExtractor(HTMLParser):
29
+ def __init__(self):
30
+ super().__init__()
31
+ self.in_script_or_style = False
32
+ self.title = ""
33
+ self.in_title = False
34
+ self.text_chunks = []
35
+ self.headings = []
36
+
37
+ def handle_starttag(self, tag, attrs):
38
+ if tag in ("script", "style", "noscript", "svg", "header", "footer", "nav"):
39
+ self.in_script_or_style = True
40
+ elif tag == "title":
41
+ self.in_title = True
42
+ elif tag in ("h1", "h2", "h3", "h4", "h5", "h6"):
43
+ self.text_chunks.append(f"\n\n{'#' * int(tag[1])} ")
44
+ elif tag in ("p", "div", "article", "section"):
45
+ self.text_chunks.append("\n\n")
46
+ elif tag == "li":
47
+ self.text_chunks.append("\n- ")
48
+ elif tag == "br":
49
+ self.text_chunks.append("\n")
50
+
51
+ def handle_endtag(self, tag):
52
+ if tag in ("script", "style", "noscript", "svg", "header", "footer", "nav"):
53
+ self.in_script_or_style = False
54
+ elif tag == "title":
55
+ self.in_title = False
56
+ elif tag in ("p", "div", "article", "section"):
57
+ self.text_chunks.append("\n")
58
+
59
+ def handle_data(self, data):
60
+ if self.in_script_or_style:
61
+ return
62
+ if self.in_title:
63
+ self.title += data
64
+ else:
65
+ self.text_chunks.append(data)
66
+
67
+ def get_markdown(self) -> str:
68
+ raw_text = "".join(self.text_chunks)
69
+ # Clean multiple blank lines
70
+ clean_text = re.sub(r"\n{3,}", "\n\n", raw_text).strip()
71
+ return clean_text
72
+
73
+ def fetch_and_cache(url: str, research_dir: str = ".research") -> str:
74
+ headers = {
75
+ "User-Agent": USER_AGENT,
76
+ "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
77
+ "Accept-Language": "en-US,en;q=0.5",
78
+ }
79
+ req = urllib.request.Request(url, headers=headers)
80
+ try:
81
+ with urllib.request.urlopen(req, timeout=20) as resp:
82
+ content_type = resp.headers.get("Content-Type", "")
83
+ raw_bytes = resp.read()
84
+ html = raw_bytes.decode("utf-8", errors="replace")
85
+ except Exception as e:
86
+ sys.stderr.write(f"[epistemic-fetch] Error fetching URL {url}: {e}\n")
87
+ return f"Error fetching {url}: {e}"
88
+
89
+ parser = HTMLToMarkdownExtractor()
90
+ parser.feed(html)
91
+ page_title = parser.title.strip() or url
92
+ body_markdown = parser.get_markdown()
93
+
94
+ # Cache into .research/sources/<sha256>.md
95
+ sha256_hash = "uncached"
96
+ cache_file = "not saved"
97
+ if SourceHasher:
98
+ from pathlib import Path
99
+ hasher = SourceHasher(base_dir=Path(research_dir))
100
+ sha256_hash = hasher.store_source(
101
+ url=url,
102
+ title=page_title,
103
+ content=body_markdown
104
+ )
105
+ cache_file = str(hasher.sources_dir / f"{sha256_hash}.md")
106
+ else:
107
+ import hashlib
108
+ sha256_hash = hashlib.sha256(body_markdown.encode("utf-8")).hexdigest()
109
+
110
+ output = []
111
+ output.append(f"<!-- EPISTEMIC_SOURCE_HASH: {sha256_hash} -->")
112
+ output.append(f"<!-- CACHED_AT: {cache_file} -->")
113
+ output.append(f"# {page_title}\n")
114
+ output.append(f"**Source URL:** {url}")
115
+ output.append(f"**Content SHA-256:** `{sha256_hash}`")
116
+ output.append(f"**Verification Tag:** `[VERIFIED: {sha256_hash[:16]}]`\n")
117
+ output.append(body_markdown)
118
+
119
+ return "\n".join(output)
120
+
121
+ def main():
122
+ url = ""
123
+ research_dir = ".research"
124
+
125
+ if not sys.stdin.isatty():
126
+ try:
127
+ stdin_data = sys.stdin.read().strip()
128
+ if stdin_data:
129
+ try:
130
+ payload = json.loads(stdin_data)
131
+ url = payload.get("url", "")
132
+ research_dir = payload.get("research_dir", research_dir)
133
+ except json.JSONDecodeError:
134
+ url = stdin_data
135
+ except Exception:
136
+ pass
137
+
138
+ args = sys.argv[1:]
139
+ idx = 0
140
+ while idx < len(args):
141
+ arg = args[idx]
142
+ if arg == "--dir" and idx + 1 < len(args):
143
+ research_dir = args[idx + 1]
144
+ idx += 1
145
+ elif not url and not arg.startswith("--"):
146
+ url = arg
147
+ idx += 1
148
+
149
+ if not url:
150
+ sys.stderr.write("Usage: fetch.py [options] <url>\nOr pipe JSON: echo '{\"url\":\"https://...\"}' | fetch.py\n")
151
+ sys.exit(1)
152
+
153
+ result = fetch_and_cache(url=url, research_dir=research_dir)
154
+ print(result)
155
+
156
+ if __name__ == "__main__":
157
+ main()
@@ -0,0 +1,162 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ Zero-API-Key Web Search Engine for IUMBTEMS.
4
+ DuckDuckGo HTML/lite parser with domain filtering and Claude Code XML formatting.
5
+ """
6
+
7
+ import sys
8
+ import json
9
+ import re
10
+ import urllib.request
11
+ import urllib.parse
12
+ from typing import List, Dict, Optional
13
+
14
+ USER_AGENT = "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
15
+
16
+ def search_duckduckgo(
17
+ query: str,
18
+ allowed_domains: Optional[List[str]] = None,
19
+ blocked_domains: Optional[List[str]] = None,
20
+ max_results: int = 10
21
+ ) -> List[Dict[str, str]]:
22
+ """Execute search query against DuckDuckGo Lite without API keys."""
23
+ effective_query = query.strip()
24
+ if allowed_domains:
25
+ domain_filters = " " + " OR ".join(f"site:{d}" for d in allowed_domains)
26
+ effective_query += domain_filters
27
+ if blocked_domains:
28
+ domain_filters = " " + " ".join(f"-site:{d}" for d in blocked_domains)
29
+ effective_query += domain_filters
30
+
31
+ url = "https://lite.duckduckgo.com/lite/"
32
+ data = urllib.parse.urlencode({"q": effective_query}).encode("utf-8")
33
+ headers = {
34
+ "User-Agent": USER_AGENT,
35
+ "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
36
+ "Accept-Language": "en-US,en;q=0.5",
37
+ "Referer": "https://lite.duckduckgo.com/",
38
+ }
39
+
40
+ req = urllib.request.Request(url, data=data, headers=headers)
41
+ try:
42
+ with urllib.request.urlopen(req, timeout=15) as resp:
43
+ html = resp.read().decode("utf-8", errors="replace")
44
+ except Exception as e:
45
+ sys.stderr.write(f"[epistemic-search] Network error fetching search results: {e}\n")
46
+ return []
47
+
48
+ # Parse result links and snippets from DuckDuckGo Lite table rows
49
+ links = re.findall(
50
+ r"<a[^>]*href=['\"]([^'\"]+)['\"][^>]*class=['\"]result-link['\"][^>]*>(.*?)</a>",
51
+ html,
52
+ re.DOTALL
53
+ )
54
+ snippets = re.findall(
55
+ r"<td[^>]+class=['\"]result-snippet['\"][^>]*>(.*?)</td>",
56
+ html,
57
+ re.DOTALL
58
+ )
59
+
60
+ results = []
61
+ for (href, title_html), snip_html in zip(links, snippets):
62
+ # Unwrap DuckDuckGo redirect if present
63
+ m = re.search(r"uddg=([^&]+)", href)
64
+ if m:
65
+ clean_url = urllib.parse.unquote(m.group(1))
66
+ else:
67
+ clean_url = href
68
+
69
+ if not clean_url.startswith("http"):
70
+ continue
71
+
72
+ clean_title = re.sub(r"<[^>]+>", "", title_html).strip()
73
+ clean_snippet = re.sub(r"<[^>]+>", "", snip_html).strip()
74
+
75
+ # Check domain filtering manually in case DDG syntax didn't catch all
76
+ parsed_url = urllib.parse.urlparse(clean_url)
77
+ hostname = parsed_url.hostname or ""
78
+ if allowed_domains and not any(hostname == d or hostname.endswith("." + d) for d in allowed_domains):
79
+ continue
80
+ if blocked_domains and any(hostname == d or hostname.endswith("." + d) for d in blocked_domains):
81
+ continue
82
+
83
+ results.append({
84
+ "title": clean_title,
85
+ "url": clean_url,
86
+ "snippet": clean_snippet
87
+ })
88
+
89
+ if len(results) >= max_results:
90
+ break
91
+
92
+ return results
93
+
94
+ def format_xml(results: List[Dict[str, str]]) -> str:
95
+ """Format results as Claude Code <search_results> XML."""
96
+ lines = ["<search_results>"]
97
+ for r in results:
98
+ lines.append(" <result>")
99
+ lines.append(f" <title>{urllib.parse.quote(r['title'])}</title>")
100
+ lines.append(f" <url>{r['url']}</url>")
101
+ lines.append(f" <snippet>{r['snippet']}</snippet>")
102
+ lines.append(" </result>")
103
+ lines.append("</search_results>")
104
+ return "\n".join(lines)
105
+
106
+ def main():
107
+ query = ""
108
+ allowed_domains = None
109
+ blocked_domains = None
110
+ output_json = False
111
+
112
+ # Check stdin first
113
+ if not sys.stdin.isatty():
114
+ try:
115
+ stdin_data = sys.stdin.read().strip()
116
+ if stdin_data:
117
+ try:
118
+ payload = json.loads(stdin_data)
119
+ query = payload.get("query", "")
120
+ allowed_domains = payload.get("allowed_domains")
121
+ blocked_domains = payload.get("blocked_domains")
122
+ except json.JSONDecodeError:
123
+ query = stdin_data
124
+ except Exception:
125
+ pass
126
+
127
+ # Command line args override
128
+ args = sys.argv[1:]
129
+ idx = 0
130
+ while idx < len(args):
131
+ arg = args[idx]
132
+ if arg == "--json":
133
+ output_json = True
134
+ elif arg == "--allowed-domain" and idx + 1 < len(args):
135
+ allowed_domains = allowed_domains or []
136
+ allowed_domains.append(args[idx + 1])
137
+ idx += 1
138
+ elif arg == "--blocked-domain" and idx + 1 < len(args):
139
+ blocked_domains = blocked_domains or []
140
+ blocked_domains.append(args[idx + 1])
141
+ idx += 1
142
+ elif not query and not arg.startswith("--"):
143
+ query = arg
144
+ idx += 1
145
+
146
+ if not query:
147
+ sys.stderr.write("Usage: search.py [options] <query>\nOr pipe JSON: echo '{\"query\":\"...\"}' | search.py\n")
148
+ sys.exit(1)
149
+
150
+ results = search_duckduckgo(
151
+ query=query,
152
+ allowed_domains=allowed_domains,
153
+ blocked_domains=blocked_domains
154
+ )
155
+
156
+ if output_json:
157
+ print(json.dumps(results, indent=2))
158
+ else:
159
+ print(format_xml(results))
160
+
161
+ if __name__ == "__main__":
162
+ main()
@@ -0,0 +1,49 @@
1
+ ---
2
+ name: oss-scout
3
+ description: Open-source software discovery, dependency vetting, and clean-room implementation scouting. Evaluates GitHub repositories, package ecosystems, licenses, and architecture to discover code to adopt or borrow.
4
+ ---
5
+
6
+ # Open Source Scout & Clean-Room Harvesting Engine
7
+
8
+ Scout the open-source software ecosystem to discover mature libraries, reference implementations, and algorithms for research-based development.
9
+ The scout deploys a **Discovery Scout (Thesis)** to find high-performance implementations and a **Licensing & Bloat Red-Teamer (Antithesis)** to protect your project against viral copyleft, unmaintained abandonware, and security CVEs.
10
+
11
+ ## 1. Invoking an Open Source Scout
12
+
13
+ Search for open-source solutions to implement a feature:
14
+ ```bash
15
+ iumbtems scout "Find high-throughput zero-dependency Raft consensus implementations in Rust or Go"
16
+ ```
17
+ Or via npx:
18
+ ```bash
19
+ npx @heretek-ai/epistemic-swarm scout "Scout vector database indexing algorithms with MIT or Apache-2.0 license"
20
+ ```
21
+ Or directly with the python runner:
22
+ ```bash
23
+ python3 runner/research_swarm.py --mode scout --objective "Explore clean-room alternatives to AGPL licensed search engines"
24
+ ```
25
+
26
+ ## 2. Dialectic Evaluation Workflow
27
+
28
+ 1. **Discovery (Alpha)**
29
+ - Explores GitHub, GitLab, crates.io, PyPI, npm, and Go packages.
30
+ - Compares star velocity, release frequency, benchmark throughput, and API design.
31
+ - Automatically caches repository documentation into `.research/sources/<sha256>.md`.
32
+
33
+ 2. **Adversarial Red-Teaming (Beta)**
34
+ - **License Contamination**: Flags AGPL/GPL requirements that could compromise proprietary or permissive codebases.
35
+ - **Maintenance Health**: Detects abandonware, stagnant commit logs, unresponsive PR queues, and solo maintainer risks.
36
+ - **Dependency Weight**: Audits transitive dependency explosion and bundle size.
37
+ - **Security Surface**: Scans for known CVEs and malicious package takeover vulnerabilities.
38
+
39
+ 3. **Synthesis & Clean-Room Blueprints**
40
+ - Synthesizes findings into `.research/oss_scout_report.md` and `.research/oss_scout_dossier.json`.
41
+ - Produces a **Clean-Room Blueprint**: an algorithmic breakdown allowing in-tree implementation of the core feature without licensing entanglement.
42
+
43
+ ## 3. Supported Package Ecosystems
44
+ - GitHub & GitLab Repositories
45
+ - Rust (crates.io)
46
+ - Python (PyPI)
47
+ - TypeScript / Node.js (npm)
48
+ - Go Modules
49
+ - C / C++ header-only libraries
@@ -0,0 +1,36 @@
1
+ ---
2
+ name: research-cache
3
+ description: Content-addressed document caching and quote verification skill. Hashes retrieved web pages and academic papers to SHA-256 for mathematical auditability.
4
+ ---
5
+
6
+ # Content-Addressed Research Cache & Verification
7
+
8
+ To maintain epistemic integrity, every document fetched from the web, arXiv, or technical docs must be cached locally with a content-addressed SHA-256 fingerprint before its claims can be cited.
9
+
10
+ ## 1. CACHING A SOURCE
11
+ When you fetch or scrape a URL:
12
+ ```bash
13
+ python3 skills/research-cache/hasher.py cache \
14
+ --url "https://arxiv.org/abs/2407.21783" \
15
+ --title "Llama 3 Herd of Models" \
16
+ --content "$(cat fetched_paper.md)"
17
+ ```
18
+ This prints the content hash:
19
+ ```
20
+ [CACHED] 3f8a9e21... -> .research/sources/3f8a9e21....md
21
+ ```
22
+
23
+ ## 2. CITING WITH HASHES
24
+ In your dossiers and markdown reports, cite the claim using the hash:
25
+ `[VERIFIED: 3f8a9e21]`
26
+
27
+ Ensure that any `verbatim_quote` you provide is an exact substring from the cached markdown document.
28
+
29
+ ## 3. AUDITING A QUOTE
30
+ The Epistemic Auditor verifies claims using:
31
+ ```bash
32
+ python3 skills/research-cache/hasher.py verify \
33
+ --hash "3f8a9e21..." \
34
+ --quote "Our FPGA pipeline executes the Poseidon round constraints in 184ms"
35
+ ```
36
+ If the quote does not match, the claim is rejected and flagged as unverified.
File without changes
@@ -0,0 +1,195 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ Content-Addressed Research Cache & Verbatim Quote Verification Engine.
4
+ Handles SHA-256 document hashing, metadata indexing, and substring auditing.
5
+ """
6
+
7
+ import os
8
+ import sys
9
+ import json
10
+ import hashlib
11
+ import re
12
+ import argparse
13
+ from datetime import datetime, timezone
14
+ from pathlib import Path
15
+ from typing import Dict, Any, Optional, Tuple
16
+
17
+ class SourceHasher:
18
+ def __init__(self, base_dir: Optional[Path] = None):
19
+ self.base_dir = base_dir or Path(".research")
20
+ self.sources_dir = self.base_dir / "sources"
21
+ self.sources_dir.mkdir(parents=True, exist_ok=True)
22
+
23
+ @staticmethod
24
+ def compute_sha256(content: str) -> str:
25
+ """Compute standard hex SHA-256 hash of normalized UTF-8 string."""
26
+ normalized = content.strip().encode("utf-8")
27
+ return hashlib.sha256(normalized).hexdigest()
28
+
29
+ def store_source(self, url: str, content: str, title: Optional[str] = None,
30
+ tier: str = "WEB_DOCUMENT", metadata: Optional[Dict[str, Any]] = None) -> str:
31
+ """Store content and metadata content-addressed by SHA-256."""
32
+ content_hash = self.compute_sha256(content)
33
+ md_path = self.sources_dir / f"{content_hash}.md"
34
+ json_path = self.sources_dir / f"{content_hash}.json"
35
+
36
+ # Write clean markdown
37
+ with open(md_path, "w", encoding="utf-8") as f:
38
+ f.write(content)
39
+
40
+ # Write metadata
41
+ meta = {
42
+ "hash": content_hash,
43
+ "url": url,
44
+ "title": title or "Untitled Source",
45
+ "tier": tier,
46
+ "cached_at": datetime.now(timezone.utc).isoformat(),
47
+ "byte_size": len(content.encode("utf-8")),
48
+ "char_count": len(content),
49
+ "custom_metadata": metadata or {}
50
+ }
51
+ with open(json_path, "w", encoding="utf-8") as f:
52
+ json.dump(meta, f, indent=2)
53
+
54
+ return content_hash
55
+
56
+ def get_source_content(self, content_hash: str) -> Optional[str]:
57
+ """Retrieve stored markdown content by full or prefix hash."""
58
+ target_file = self._resolve_hash_file(content_hash, ".md")
59
+ if target_file and target_file.exists():
60
+ with open(target_file, "r", encoding="utf-8") as f:
61
+ return f.read()
62
+ return None
63
+
64
+ def get_source_metadata(self, content_hash: str) -> Optional[Dict[str, Any]]:
65
+ """Retrieve stored metadata by full or prefix hash."""
66
+ target_file = self._resolve_hash_file(content_hash, ".json")
67
+ if target_file and target_file.exists():
68
+ with open(target_file, "r", encoding="utf-8") as f:
69
+ return json.load(f)
70
+ return None
71
+
72
+ def _resolve_hash_file(self, hash_prefix: str, extension: str) -> Optional[Path]:
73
+ """Resolves exact or prefix hash to disk path."""
74
+ exact_path = self.sources_dir / f"{hash_prefix}{extension}"
75
+ if exact_path.exists():
76
+ return exact_path
77
+
78
+ # If prefix match
79
+ matches = list(self.sources_dir.glob(f"{hash_prefix}*{extension}"))
80
+ if len(matches) == 1:
81
+ return matches[0]
82
+ return None
83
+
84
+ @staticmethod
85
+ def normalize_text_for_search(text: str) -> str:
86
+ """Collapse whitespace and normalize typography for substring matching."""
87
+ # Replace smart quotes and dashes
88
+ text = text.replace("“", '"').replace("”", '"').replace("‘", "'").replace("’", "'")
89
+ text = text.replace("—", "-").replace("–", "-")
90
+ # Collapse all whitespace to single spaces
91
+ return re.sub(r"\s+", " ", text).strip().lower()
92
+
93
+ def verify_quote(self, content_hash: str, quote: str) -> Tuple[bool, float, Optional[str]]:
94
+ """
95
+ Verifies whether quote exists in cached document.
96
+ Returns: (is_verified, confidence_score, context_match)
97
+ """
98
+ source_text = self.get_source_content(content_hash)
99
+ if not source_text:
100
+ return False, 0.0, f"Source hash '{content_hash}' not found in cache."
101
+
102
+ # 1. Exact match test
103
+ if quote.strip() in source_text:
104
+ return True, 1.0, "Exact substring match found."
105
+
106
+ # 2. Normalized whitespace match test
107
+ norm_source = self.normalize_text_for_search(source_text)
108
+ norm_quote = self.normalize_text_for_search(quote)
109
+ if norm_quote in norm_source:
110
+ return True, 0.98, "Normalized whitespace match found."
111
+
112
+ # 3. Sliding window token overlap test
113
+ quote_words = norm_quote.split()
114
+ if len(quote_words) < 4:
115
+ return False, 0.0, "Quote too short and not found verbatim."
116
+
117
+ window_size = len(quote_words)
118
+ source_words = norm_source.split()
119
+ best_score = 0.0
120
+
121
+ for i in range(max(1, len(source_words) - window_size + 1)):
122
+ window = source_words[i:i + window_size]
123
+ matches = sum(1 for w1, w2 in zip(quote_words, window) if w1 == w2)
124
+ score = matches / window_size
125
+ if score > best_score:
126
+ best_score = score
127
+ if best_score >= 0.90:
128
+ break
129
+
130
+ if best_score >= 0.88:
131
+ return True, best_score, f"High-confidence fuzzy match ({best_score:.2f})."
132
+
133
+ return False, best_score, f"Verification failed. Highest word overlap: {best_score:.2f}."
134
+
135
+
136
+ def main():
137
+ parser = argparse.ArgumentParser(description="Epistemic Swarm Content Hasher & Quote Verifier")
138
+ subparsers = parser.add_subparsers(dest="command")
139
+
140
+ # Cache command
141
+ cache_parser = subparsers.add_parser("cache", help="Cache a document")
142
+ cache_parser.add_argument("--url", required=True, help="Original URL")
143
+ cache_parser.add_argument("--title", default="Untitled", help="Document Title")
144
+ cache_parser.add_argument("--tier", default="WEB_DOCUMENT", help="Source tier")
145
+ cache_parser.add_argument("--content", help="Raw text content (or read from stdin)")
146
+ cache_parser.add_argument("--dir", default=".research", help="Base .research directory")
147
+
148
+ # Verify command
149
+ verify_parser = subparsers.add_parser("verify", help="Verify a verbatim quote")
150
+ verify_parser.add_argument("--hash", required=True, help="Document SHA-256 hash")
151
+ verify_parser.add_argument("--quote", required=True, help="Verbatim quote to check")
152
+ verify_parser.add_argument("--dir", default=".research", help="Base .research directory")
153
+
154
+ # List command
155
+ list_parser = subparsers.add_parser("list", help="List cached sources")
156
+ list_parser.add_argument("--dir", default=".research", help="Base .research directory")
157
+
158
+ args = parser.parse_args()
159
+ hasher = SourceHasher(base_dir=Path(args.dir if hasattr(args, "dir") else ".research"))
160
+
161
+ if args.command == "cache":
162
+ content = args.content
163
+ if not content:
164
+ if not sys.stdin.isatty():
165
+ content = sys.stdin.read()
166
+ else:
167
+ print("Error: No content provided via --content or stdin.", file=sys.stderr)
168
+ sys.exit(1)
169
+ h = hasher.store_source(url=args.url, content=content, title=args.title, tier=args.tier)
170
+ print(f"[CACHED] {h} -> {args.title} ({args.url})")
171
+
172
+ elif args.command == "verify":
173
+ verified, conf, msg = hasher.verify_quote(content_hash=args.hash, quote=args.quote)
174
+ result = {
175
+ "hash": args.hash,
176
+ "verified": verified,
177
+ "confidence": conf,
178
+ "message": msg
179
+ }
180
+ print(json.dumps(result, indent=2))
181
+ sys.exit(0 if verified else 1)
182
+
183
+ elif args.command == "list":
184
+ sources = list(hasher.sources_dir.glob("*.json"))
185
+ print(f"Total Cached Sources: {len(sources)}")
186
+ for p in sources:
187
+ with open(p, "r", encoding="utf-8") as f:
188
+ d = json.load(f)
189
+ print(f"- [{d['hash'][:10]}...] {d['title']} ({d['url']})")
190
+ else:
191
+ parser.print_help()
192
+
193
+
194
+ if __name__ == "__main__":
195
+ main()
@@ -0,0 +1,72 @@
1
+ ---
2
+ name: swarm-config
3
+ description: Dynamic configuration and settings skill for the Epistemic Swarm research harness. Manage search engines, research depth, dialectic iterations, operating modes (research, audit, scout), and license filters.
4
+ ---
5
+
6
+ # Epistemic Swarm Configuration & Parameter Tuning
7
+
8
+ Use this skill to inspect, tune, and persist research parameters into `.research/config.json`.
9
+ Settings are automatically loaded by the Swarm Runner, Epistemic Auditor, and CLI.
10
+
11
+ ## 1. Interactive Configuration
12
+
13
+ To launch the interactive configuration prompt:
14
+ ```bash
15
+ python3 skills/swarm_config/configure.py --interactive
16
+ ```
17
+
18
+ This guides you through selecting:
19
+ 1. **Primary Search Engine**:
20
+ - `duckduckgo` (Default, zero API key required, completely private)
21
+ - `brave` (Brave Search API for high-precision SERP results)
22
+ - `firecrawl` (Deep web scraping & JavaScript rendering)
23
+ - `searxng` (Self-hosted privacy metasearch aggregator)
24
+ 2. **Research Depth & Dialectic Iterations**:
25
+ - `1` (Rapid brief, minimal token usage)
26
+ - `2` (Standard thesis vs. antithesis dialectic - recommended)
27
+ - `3` (Deep multi-pass verification)
28
+ - `4` (Exhaustive multi-scope investigation)
29
+ 3. **Operating Mode**:
30
+ - `research` (Empirical literature & web synthesis)
31
+ - `audit` (Deep codebase architecture, security, and vulnerability red-teaming)
32
+ - `scout` (Open-source software discovery & clean-room harvesting)
33
+ - `hybrid` (Combined codebase audit + web research)
34
+
35
+ ## 2. Direct CLI Configuration
36
+
37
+ Inspect the current active configuration:
38
+ ```bash
39
+ python3 skills/swarm_config/configure.py --show
40
+ ```
41
+ or via CLI:
42
+ ```bash
43
+ iumbtems config
44
+ ```
45
+
46
+ Update parameters directly via flags:
47
+ ```bash
48
+ # Set search engine to duckduckgo and depth to 3
49
+ python3 skills/swarm_config/configure.py --engine duckduckgo --depth 3
50
+
51
+ # Set operating mode to codebase audit
52
+ python3 skills/swarm_config/configure.py --mode audit
53
+
54
+ # Set divergence threshold
55
+ python3 skills/swarm_config/configure.py --divergence 0.8
56
+ ```
57
+
58
+ ## 3. Configuration Schema (`.research/config.json`)
59
+
60
+ The active project configuration is persisted at `.research/config.json`:
61
+ ```json
62
+ {
63
+ "search_engine": "duckduckgo",
64
+ "max_iterations": 2,
65
+ "divergence_threshold": 0.75,
66
+ "mode": "research",
67
+ "cache_raw_markdown": true,
68
+ "license_whitelist": ["MIT", "Apache-2.0", "BSD-3-Clause", "ISC"],
69
+ "output_dir": ".research"
70
+ }
71
+ ```
72
+ All swarm agents read this file at initialization time.
@@ -0,0 +1,4 @@
1
+ """Epistemic Swarm configuration package."""
2
+ from .configure import load_config, save_config, DEFAULT_CONFIG
3
+
4
+ __all__ = ["load_config", "save_config", "DEFAULT_CONFIG"]