@heretek-ai/epistemic-swarm 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +14 -0
- package/.claude-plugin/plugin.json +58 -0
- package/LICENSE +126 -0
- package/README.md +130 -0
- package/bin/cli.js +119 -0
- package/config/claude-settings-patch.json +10 -0
- package/config/docker-compose.infra.yml +23 -0
- package/config/mcp-research-servers.json +26 -0
- package/config/searxng_mcp.py +133 -0
- package/install.sh +102 -0
- package/package.json +52 -0
- package/prompts/agent_alpha_thesis.md +70 -0
- package/prompts/agent_beta_antithesis.md +78 -0
- package/prompts/base_epistemic_system.md +50 -0
- package/prompts/epistemic_auditor.md +74 -0
- package/prompts/orchestrator.md +70 -0
- package/runner/__init__.py +0 -0
- package/runner/__pycache__/__init__.cpython-314.pyc +0 -0
- package/runner/__pycache__/auditor_engine.cpython-314.pyc +0 -0
- package/runner/__pycache__/research_swarm.cpython-314.pyc +0 -0
- package/runner/__pycache__/state_machine.cpython-314.pyc +0 -0
- package/runner/auditor_engine.py +222 -0
- package/runner/research_swarm.py +337 -0
- package/runner/state_machine.py +192 -0
- package/runner/tests/__pycache__/test_swarm.cpython-314.pyc +0 -0
- package/runner/tests/test_swarm.py +178 -0
- package/skills/grilling/SKILL.md +48 -0
- package/skills/grilling/__init__.py +0 -0
- package/skills/grilling/socratic_tree.py +148 -0
- package/skills/research-cache/SKILL.md +36 -0
- package/skills/research-cache/__init__.py +0 -0
- package/skills/research-cache/__pycache__/__init__.cpython-314.pyc +0 -0
- package/skills/research-cache/__pycache__/hasher.cpython-314.pyc +0 -0
- package/skills/research-cache/hasher.py +195 -0
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Content-Addressed Research Cache & Verbatim Quote Verification Engine.
|
|
4
|
+
Handles SHA-256 document hashing, metadata indexing, and substring auditing.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import os
|
|
8
|
+
import sys
|
|
9
|
+
import json
|
|
10
|
+
import hashlib
|
|
11
|
+
import re
|
|
12
|
+
import argparse
|
|
13
|
+
from datetime import datetime, timezone
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Dict, Any, Optional, Tuple
|
|
16
|
+
|
|
17
|
+
class SourceHasher:
|
|
18
|
+
def __init__(self, base_dir: Optional[Path] = None):
|
|
19
|
+
self.base_dir = base_dir or Path(".research")
|
|
20
|
+
self.sources_dir = self.base_dir / "sources"
|
|
21
|
+
self.sources_dir.mkdir(parents=True, exist_ok=True)
|
|
22
|
+
|
|
23
|
+
@staticmethod
|
|
24
|
+
def compute_sha256(content: str) -> str:
|
|
25
|
+
"""Compute standard hex SHA-256 hash of normalized UTF-8 string."""
|
|
26
|
+
normalized = content.strip().encode("utf-8")
|
|
27
|
+
return hashlib.sha256(normalized).hexdigest()
|
|
28
|
+
|
|
29
|
+
def store_source(self, url: str, content: str, title: Optional[str] = None,
|
|
30
|
+
tier: str = "WEB_DOCUMENT", metadata: Optional[Dict[str, Any]] = None) -> str:
|
|
31
|
+
"""Store content and metadata content-addressed by SHA-256."""
|
|
32
|
+
content_hash = self.compute_sha256(content)
|
|
33
|
+
md_path = self.sources_dir / f"{content_hash}.md"
|
|
34
|
+
json_path = self.sources_dir / f"{content_hash}.json"
|
|
35
|
+
|
|
36
|
+
# Write clean markdown
|
|
37
|
+
with open(md_path, "w", encoding="utf-8") as f:
|
|
38
|
+
f.write(content)
|
|
39
|
+
|
|
40
|
+
# Write metadata
|
|
41
|
+
meta = {
|
|
42
|
+
"hash": content_hash,
|
|
43
|
+
"url": url,
|
|
44
|
+
"title": title or "Untitled Source",
|
|
45
|
+
"tier": tier,
|
|
46
|
+
"cached_at": datetime.now(timezone.utc).isoformat(),
|
|
47
|
+
"byte_size": len(content.encode("utf-8")),
|
|
48
|
+
"char_count": len(content),
|
|
49
|
+
"custom_metadata": metadata or {}
|
|
50
|
+
}
|
|
51
|
+
with open(json_path, "w", encoding="utf-8") as f:
|
|
52
|
+
json.dump(meta, f, indent=2)
|
|
53
|
+
|
|
54
|
+
return content_hash
|
|
55
|
+
|
|
56
|
+
def get_source_content(self, content_hash: str) -> Optional[str]:
|
|
57
|
+
"""Retrieve stored markdown content by full or prefix hash."""
|
|
58
|
+
target_file = self._resolve_hash_file(content_hash, ".md")
|
|
59
|
+
if target_file and target_file.exists():
|
|
60
|
+
with open(target_file, "r", encoding="utf-8") as f:
|
|
61
|
+
return f.read()
|
|
62
|
+
return None
|
|
63
|
+
|
|
64
|
+
def get_source_metadata(self, content_hash: str) -> Optional[Dict[str, Any]]:
|
|
65
|
+
"""Retrieve stored metadata by full or prefix hash."""
|
|
66
|
+
target_file = self._resolve_hash_file(content_hash, ".json")
|
|
67
|
+
if target_file and target_file.exists():
|
|
68
|
+
with open(target_file, "r", encoding="utf-8") as f:
|
|
69
|
+
return json.load(f)
|
|
70
|
+
return None
|
|
71
|
+
|
|
72
|
+
def _resolve_hash_file(self, hash_prefix: str, extension: str) -> Optional[Path]:
|
|
73
|
+
"""Resolves exact or prefix hash to disk path."""
|
|
74
|
+
exact_path = self.sources_dir / f"{hash_prefix}{extension}"
|
|
75
|
+
if exact_path.exists():
|
|
76
|
+
return exact_path
|
|
77
|
+
|
|
78
|
+
# If prefix match
|
|
79
|
+
matches = list(self.sources_dir.glob(f"{hash_prefix}*{extension}"))
|
|
80
|
+
if len(matches) == 1:
|
|
81
|
+
return matches[0]
|
|
82
|
+
return None
|
|
83
|
+
|
|
84
|
+
@staticmethod
|
|
85
|
+
def normalize_text_for_search(text: str) -> str:
|
|
86
|
+
"""Collapse whitespace and normalize typography for substring matching."""
|
|
87
|
+
# Replace smart quotes and dashes
|
|
88
|
+
text = text.replace("“", '"').replace("”", '"').replace("‘", "'").replace("’", "'")
|
|
89
|
+
text = text.replace("—", "-").replace("–", "-")
|
|
90
|
+
# Collapse all whitespace to single spaces
|
|
91
|
+
return re.sub(r"\s+", " ", text).strip().lower()
|
|
92
|
+
|
|
93
|
+
def verify_quote(self, content_hash: str, quote: str) -> Tuple[bool, float, Optional[str]]:
|
|
94
|
+
"""
|
|
95
|
+
Verifies whether quote exists in cached document.
|
|
96
|
+
Returns: (is_verified, confidence_score, context_match)
|
|
97
|
+
"""
|
|
98
|
+
source_text = self.get_source_content(content_hash)
|
|
99
|
+
if not source_text:
|
|
100
|
+
return False, 0.0, f"Source hash '{content_hash}' not found in cache."
|
|
101
|
+
|
|
102
|
+
# 1. Exact match test
|
|
103
|
+
if quote.strip() in source_text:
|
|
104
|
+
return True, 1.0, "Exact substring match found."
|
|
105
|
+
|
|
106
|
+
# 2. Normalized whitespace match test
|
|
107
|
+
norm_source = self.normalize_text_for_search(source_text)
|
|
108
|
+
norm_quote = self.normalize_text_for_search(quote)
|
|
109
|
+
if norm_quote in norm_source:
|
|
110
|
+
return True, 0.98, "Normalized whitespace match found."
|
|
111
|
+
|
|
112
|
+
# 3. Sliding window token overlap test
|
|
113
|
+
quote_words = norm_quote.split()
|
|
114
|
+
if len(quote_words) < 4:
|
|
115
|
+
return False, 0.0, "Quote too short and not found verbatim."
|
|
116
|
+
|
|
117
|
+
window_size = len(quote_words)
|
|
118
|
+
source_words = norm_source.split()
|
|
119
|
+
best_score = 0.0
|
|
120
|
+
|
|
121
|
+
for i in range(max(1, len(source_words) - window_size + 1)):
|
|
122
|
+
window = source_words[i:i + window_size]
|
|
123
|
+
matches = sum(1 for w1, w2 in zip(quote_words, window) if w1 == w2)
|
|
124
|
+
score = matches / window_size
|
|
125
|
+
if score > best_score:
|
|
126
|
+
best_score = score
|
|
127
|
+
if best_score >= 0.90:
|
|
128
|
+
break
|
|
129
|
+
|
|
130
|
+
if best_score >= 0.88:
|
|
131
|
+
return True, best_score, f"High-confidence fuzzy match ({best_score:.2f})."
|
|
132
|
+
|
|
133
|
+
return False, best_score, f"Verification failed. Highest word overlap: {best_score:.2f}."
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def main():
|
|
137
|
+
parser = argparse.ArgumentParser(description="Epistemic Swarm Content Hasher & Quote Verifier")
|
|
138
|
+
subparsers = parser.add_subparsers(dest="command")
|
|
139
|
+
|
|
140
|
+
# Cache command
|
|
141
|
+
cache_parser = subparsers.add_parser("cache", help="Cache a document")
|
|
142
|
+
cache_parser.add_argument("--url", required=True, help="Original URL")
|
|
143
|
+
cache_parser.add_argument("--title", default="Untitled", help="Document Title")
|
|
144
|
+
cache_parser.add_argument("--tier", default="WEB_DOCUMENT", help="Source tier")
|
|
145
|
+
cache_parser.add_argument("--content", help="Raw text content (or read from stdin)")
|
|
146
|
+
cache_parser.add_argument("--dir", default=".research", help="Base .research directory")
|
|
147
|
+
|
|
148
|
+
# Verify command
|
|
149
|
+
verify_parser = subparsers.add_parser("verify", help="Verify a verbatim quote")
|
|
150
|
+
verify_parser.add_argument("--hash", required=True, help="Document SHA-256 hash")
|
|
151
|
+
verify_parser.add_argument("--quote", required=True, help="Verbatim quote to check")
|
|
152
|
+
verify_parser.add_argument("--dir", default=".research", help="Base .research directory")
|
|
153
|
+
|
|
154
|
+
# List command
|
|
155
|
+
list_parser = subparsers.add_parser("list", help="List cached sources")
|
|
156
|
+
list_parser.add_argument("--dir", default=".research", help="Base .research directory")
|
|
157
|
+
|
|
158
|
+
args = parser.parse_args()
|
|
159
|
+
hasher = SourceHasher(base_dir=Path(args.dir if hasattr(args, "dir") else ".research"))
|
|
160
|
+
|
|
161
|
+
if args.command == "cache":
|
|
162
|
+
content = args.content
|
|
163
|
+
if not content:
|
|
164
|
+
if not sys.stdin.isatty():
|
|
165
|
+
content = sys.stdin.read()
|
|
166
|
+
else:
|
|
167
|
+
print("Error: No content provided via --content or stdin.", file=sys.stderr)
|
|
168
|
+
sys.exit(1)
|
|
169
|
+
h = hasher.store_source(url=args.url, content=content, title=args.title, tier=args.tier)
|
|
170
|
+
print(f"[CACHED] {h} -> {args.title} ({args.url})")
|
|
171
|
+
|
|
172
|
+
elif args.command == "verify":
|
|
173
|
+
verified, conf, msg = hasher.verify_quote(content_hash=args.hash, quote=args.quote)
|
|
174
|
+
result = {
|
|
175
|
+
"hash": args.hash,
|
|
176
|
+
"verified": verified,
|
|
177
|
+
"confidence": conf,
|
|
178
|
+
"message": msg
|
|
179
|
+
}
|
|
180
|
+
print(json.dumps(result, indent=2))
|
|
181
|
+
sys.exit(0 if verified else 1)
|
|
182
|
+
|
|
183
|
+
elif args.command == "list":
|
|
184
|
+
sources = list(hasher.sources_dir.glob("*.json"))
|
|
185
|
+
print(f"Total Cached Sources: {len(sources)}")
|
|
186
|
+
for p in sources:
|
|
187
|
+
with open(p, "r", encoding="utf-8") as f:
|
|
188
|
+
d = json.load(f)
|
|
189
|
+
print(f"- [{d['hash'][:10]}...] {d['title']} ({d['url']})")
|
|
190
|
+
else:
|
|
191
|
+
parser.print_help()
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
if __name__ == "__main__":
|
|
195
|
+
main()
|