@heretek-ai/epistemic-swarm 0.4.2 → 0.4.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/hooks/hooks.json +22 -1
  3. package/package.json +1 -1
  4. package/plugins/antigravity/plugin.json +1 -1
  5. package/plugins/gemini/gemini-extension.json +1 -1
  6. package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
  7. package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
  8. package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
  9. package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
  10. package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
  11. package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
  12. package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
  13. package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
  14. package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
  15. package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
  16. package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
  17. package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
  18. package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
  19. package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
  20. package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
  21. package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
  22. package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
  23. package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
  24. package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
  25. package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
  26. package/runner/tests/__pycache__/test_opencode_ux.cpython-311.pyc +0 -0
  27. package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
  28. package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
  29. package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
  30. package/runner/tests/__pycache__/test_webcache.cpython-311.pyc +0 -0
  31. package/runner/tests/test_webcache.py +222 -0
  32. package/skills/epistemic_search/scripts/webcache.py +292 -0
  33. package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
  34. package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
  35. package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
  36. package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "epistemic-swarm",
3
- "version": "0.4.2",
3
+ "version": "0.4.3",
4
4
  "description": "High-Integrity Dialectic Research Agent Harness for Claude Code enforcing empirical evidence over parametric hallucination.",
5
5
  "author": {
6
6
  "name": "Heretek AI",
package/hooks/hooks.json CHANGED
@@ -15,7 +15,28 @@
15
15
  "hooks": [
16
16
  {
17
17
  "type": "command",
18
- "command": "echo '{\"hookSpecificOutput\":{\"hookEventName\":\"PreToolUse\",\"permissionDecision\":\"deny\",\"permissionDecisionReason\":\"WebFetch intercepted by epistemic-swarm to enforce content-addressed caching. Use epistemic fetch (python3 ${CLAUDE_PLUGIN_ROOT}/skills/epistemic_search/scripts/fetch.py \\\"<url>\\\") which automatically stores documents in .research/sources/<sha256>.md for mathematical auditability.\"}}'"
18
+ "command": "python3 \"${CLAUDE_PLUGIN_ROOT}/skills/epistemic_search/scripts/webcache.py\" gate"
19
+ }
20
+ ]
21
+ }
22
+ ],
23
+ "PostToolUse": [
24
+ {
25
+ "matcher": "WebFetch",
26
+ "hooks": [
27
+ {
28
+ "type": "command",
29
+ "command": "python3 \"${CLAUDE_PLUGIN_ROOT}/skills/epistemic_search/scripts/webcache.py\" archive"
30
+ }
31
+ ]
32
+ }
33
+ ],
34
+ "SessionStart": [
35
+ {
36
+ "hooks": [
37
+ {
38
+ "type": "command",
39
+ "command": "python3 \"${CLAUDE_PLUGIN_ROOT}/skills/epistemic_search/scripts/webcache.py\" stats"
19
40
  }
20
41
  ]
21
42
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@heretek-ai/epistemic-swarm",
3
- "version": "0.4.2",
3
+ "version": "0.4.3",
4
4
  "description": "IUMBTEMS: I Use My Brain To Express My Self — High-Integrity Dialectic Research Agent Harness for Claude Code, OpenCode V2, Pi, OMP (oh-my-pi), Gemini CLI, Codex CLI, and AntiGravity",
5
5
  "main": "bin/cli.js",
6
6
  "bin": {
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "$schema": "https://antigravity.google/schemas/v1/plugin.json",
3
3
  "name": "epistemic-swarm",
4
- "version": "0.4.2",
4
+ "version": "0.4.3",
5
5
  "description": "IUMBTEMS Epistemic Swarm: dialectic research, code audit, OSS scout, and lateral brainstorming for AntiGravity.",
6
6
  "author": "Heretek AI",
7
7
  "license": "Apache-2.0",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "epistemic-swarm",
3
- "version": "0.4.2",
3
+ "version": "0.4.3",
4
4
  "description": "IUMBTEMS Epistemic Swarm: dialectic research, code audit, OSS scout, and lateral brainstorming for Gemini CLI.",
5
5
  "mcpServers": {
6
6
  "brave-search": {
@@ -0,0 +1,222 @@
1
+ #!/usr/bin/env python3
2
+ """Tests for the WebFetch cache-through gate (webcache.py + hooks.json)."""
3
+
4
+ import json
5
+ import subprocess
6
+ import sys
7
+ import tempfile
8
+ import unittest
9
+ from datetime import datetime, timedelta, timezone
10
+ from pathlib import Path
11
+
12
+ PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
13
+ if str(PROJECT_ROOT) not in sys.path:
14
+ sys.path.insert(0, str(PROJECT_ROOT))
15
+
16
+ WEBCACHE = PROJECT_ROOT / "skills" / "epistemic_search" / "scripts" / "webcache.py"
17
+
18
+
19
+ def run_webcache(mode, payload=None, raw=None, cwd=None, env=None):
20
+ stdin = raw if raw is not None else json.dumps(payload or {})
21
+ merged = dict(__import__("os").environ)
22
+ if env:
23
+ merged.update(env)
24
+ return subprocess.run(
25
+ [sys.executable, str(WEBCACHE), mode],
26
+ input=stdin,
27
+ capture_output=True,
28
+ text=True,
29
+ cwd=str(cwd or PROJECT_ROOT),
30
+ env=merged,
31
+ )
32
+
33
+
34
+ def seed_source(sources_dir, url, content, cached_at=None):
35
+ import hashlib
36
+
37
+ digest = hashlib.sha256(content.encode("utf-8")).hexdigest()
38
+ sources_dir.mkdir(parents=True, exist_ok=True)
39
+ sources_dir.joinpath(f"{digest}.md").write_text(content, encoding="utf-8")
40
+ sources_dir.joinpath(f"{digest}.json").write_text(
41
+ json.dumps(
42
+ {
43
+ "hash": digest,
44
+ "url": url,
45
+ "title": url,
46
+ "tier": "WEB_DOCUMENT",
47
+ "cached_at": cached_at
48
+ or datetime.now(timezone.utc).isoformat(),
49
+ "byte_size": len(content.encode("utf-8")),
50
+ "char_count": len(content),
51
+ "custom_metadata": {},
52
+ }
53
+ ),
54
+ encoding="utf-8",
55
+ )
56
+ return digest
57
+
58
+
59
+ def hook_input(url, cwd):
60
+ return {"tool_name": "WebFetch", "tool_input": {"url": url}, "cwd": str(cwd)}
61
+
62
+
63
+ class TestCanonicalization(unittest.TestCase):
64
+ def test_variants_resolve_to_one_entry(self):
65
+ with tempfile.TemporaryDirectory() as tmp:
66
+ root = Path(tmp)
67
+ seed_source(
68
+ root / ".research" / "sources",
69
+ "https://example.com/p?a=1&b=2",
70
+ "hello cached",
71
+ )
72
+ for variant in [
73
+ "https://EXAMPLE.com/p?b=2&a=1#frag",
74
+ "https://example.com:443/p?a=1&b=2",
75
+ ]:
76
+ res = run_webcache("gate", hook_input(variant, root))
77
+ self.assertEqual(res.returncode, 0)
78
+ self.assertTrue(res.stdout.strip(), f"expected HIT for {variant}")
79
+ out = json.loads(res.stdout)
80
+ self.assertEqual(
81
+ out["hookSpecificOutput"]["permissionDecision"], "deny"
82
+ )
83
+ self.assertIn("hello cached", out["hookSpecificOutput"]["additionalContext"])
84
+
85
+
86
+ class TestGate(unittest.TestCase):
87
+ def test_miss_allows_silently(self):
88
+ with tempfile.TemporaryDirectory() as tmp:
89
+ res = run_webcache(
90
+ "gate", hook_input("https://example.com/new-page", Path(tmp))
91
+ )
92
+ self.assertEqual(res.returncode, 0)
93
+ self.assertEqual(res.stdout.strip(), "")
94
+
95
+ def test_hit_serves_cache(self):
96
+ with tempfile.TemporaryDirectory() as tmp:
97
+ root = Path(tmp)
98
+ digest = seed_source(
99
+ root / ".research" / "sources",
100
+ "https://code.claude.com/docs/en/hooks",
101
+ "# Hooks\n\nHook content here.",
102
+ )
103
+ res = run_webcache(
104
+ "gate", hook_input("https://code.claude.com/docs/en/hooks", root)
105
+ )
106
+ self.assertEqual(res.returncode, 0)
107
+ out = json.loads(res.stdout)
108
+ hook = out["hookSpecificOutput"]
109
+ self.assertEqual(hook["hookEventName"], "PreToolUse")
110
+ self.assertEqual(hook["permissionDecision"], "deny")
111
+ self.assertIn("Hook content here.", hook["additionalContext"])
112
+ self.assertIn(f"[VERIFIED: {digest[:16]}]", hook["additionalContext"])
113
+
114
+ def test_stale_allows_for_refresh(self):
115
+ with tempfile.TemporaryDirectory() as tmp:
116
+ root = Path(tmp)
117
+ old = (datetime.now(timezone.utc) - timedelta(days=30)).isoformat()
118
+ seed_source(
119
+ root / ".research" / "sources",
120
+ "https://example.com/aging",
121
+ "old content",
122
+ cached_at=old,
123
+ )
124
+ res = run_webcache("gate", hook_input("https://example.com/aging", root))
125
+ self.assertEqual(res.returncode, 0)
126
+ self.assertEqual(res.stdout.strip(), "")
127
+
128
+ def test_malformed_input_fails_open(self):
129
+ for raw in ["", "not json{{{", '{"tool_input": {}}']:
130
+ res = run_webcache("gate", raw=raw)
131
+ self.assertEqual(res.returncode, 0, f"raw={raw!r} stderr={res.stderr}")
132
+ self.assertEqual(res.stdout.strip(), "")
133
+
134
+
135
+ class TestArchive(unittest.TestCase):
136
+ def test_archive_writes_matching_pair(self):
137
+ with tempfile.TemporaryDirectory() as tmp:
138
+ root = Path(tmp)
139
+ payload = {
140
+ "tool_name": "WebFetch",
141
+ "tool_input": {"url": "https://example.com/live"},
142
+ "tool_response": "# Live\n\nFresh body text.",
143
+ "cwd": str(root),
144
+ }
145
+ res = run_webcache("archive", payload)
146
+ self.assertEqual(res.returncode, 0, res.stderr)
147
+ self.assertEqual(res.stdout.strip(), "")
148
+ sources = root / ".research" / "sources"
149
+ md_files = list(sources.glob("*.md"))
150
+ self.assertEqual(len(md_files), 1)
151
+ import hashlib
152
+
153
+ digest = hashlib.sha256("# Live\n\nFresh body text.".encode("utf-8")).hexdigest()
154
+ self.assertEqual(md_files[0].name, f"{digest}.md")
155
+ meta = json.loads(sources.joinpath(f"{digest}.json").read_text(encoding="utf-8"))
156
+ self.assertEqual(meta["url"], "https://example.com/live")
157
+ self.assertEqual(meta["hash"], digest)
158
+ # Idempotent: archiving again writes the same pair.
159
+ res2 = run_webcache("archive", payload)
160
+ self.assertEqual(res2.returncode, 0)
161
+ self.assertEqual(len(list(sources.glob("*.md"))), 1)
162
+
163
+ def test_archive_round_trip_gate_hit(self):
164
+ with tempfile.TemporaryDirectory() as tmp:
165
+ root = Path(tmp)
166
+ url = "https://example.com/roundtrip"
167
+ run_webcache(
168
+ "archive",
169
+ {
170
+ "tool_input": {"url": url},
171
+ "response": "Roundtrip body.",
172
+ "cwd": str(root),
173
+ },
174
+ )
175
+ res = run_webcache("gate", hook_input(url, root))
176
+ out = json.loads(res.stdout)
177
+ self.assertEqual(out["hookSpecificOutput"]["permissionDecision"], "deny")
178
+ self.assertIn("Roundtrip body.", out["hookSpecificOutput"]["additionalContext"])
179
+
180
+
181
+ class TestStats(unittest.TestCase):
182
+ def test_empty_is_silent(self):
183
+ with tempfile.TemporaryDirectory() as tmp:
184
+ res = run_webcache("stats", {"cwd": str(Path(tmp))})
185
+ self.assertEqual(res.returncode, 0)
186
+ self.assertEqual(res.stdout.strip(), "")
187
+
188
+ def test_populated_reports_count(self):
189
+ with tempfile.TemporaryDirectory() as tmp:
190
+ root = Path(tmp)
191
+ seed_source(root / ".research" / "sources", "https://a.example/1", "one")
192
+ seed_source(root / ".research" / "sources", "https://b.example/2", "two")
193
+ res = run_webcache("stats", {"cwd": str(root)})
194
+ out = json.loads(res.stdout)
195
+ self.assertIn("2 page(s) cached", out["hookSpecificOutput"]["additionalContext"])
196
+
197
+
198
+ class TestHooksManifest(unittest.TestCase):
199
+ def test_hooks_json_wires_cache_through(self):
200
+ hooks = json.loads(
201
+ (PROJECT_ROOT / "hooks" / "hooks.json").read_text(encoding="utf-8")
202
+ )
203
+ pre = hooks["hooks"]["PreToolUse"]
204
+ webfetch_pre = [m for m in pre if m.get("matcher") == "WebFetch"]
205
+ self.assertEqual(len(webfetch_pre), 1)
206
+ pre_cmd = webfetch_pre[0]["hooks"][0]["command"]
207
+ self.assertIn("webcache.py", pre_cmd)
208
+ self.assertTrue(pre_cmd.rstrip().endswith("gate"))
209
+ post = hooks["hooks"]["PostToolUse"]
210
+ webfetch_post = [m for m in post if m.get("matcher") == "WebFetch"]
211
+ self.assertEqual(len(webfetch_post), 1)
212
+ post_cmd = webfetch_post[0]["hooks"][0]["command"]
213
+ self.assertIn("webcache.py", post_cmd)
214
+ self.assertTrue(post_cmd.rstrip().endswith("archive"))
215
+ session_start = hooks["hooks"].get("SessionStart", [])
216
+ self.assertTrue(
217
+ any("webcache.py" in h.get("command", "") and h.get("command", "").rstrip().endswith("stats") for m in session_start for h in m.get("hooks", []))
218
+ )
219
+
220
+
221
+ if __name__ == "__main__":
222
+ unittest.main()
@@ -0,0 +1,292 @@
1
+ #!/usr/bin/env python3
2
+ """Content-addressed WebFetch cache-through gate for Claude Code hooks.
3
+
4
+ Replaces the blanket PreToolUse deny with the cache-through pattern
5
+ (proven by theYahia/claude-webcache):
6
+
7
+ gate (PreToolUse WebFetch): serve `.research/sources/<sha256>.md` hits
8
+ via deny + additionalContext (no network); allow misses/stale.
9
+ archive (PostToolUse WebFetch): store the live response as md + json
10
+ sidecar, same layout as SourceHasher (content-hash filename).
11
+ stats (SessionStart): one-line cache summary via additionalContext.
12
+
13
+ Fail-open everywhere: any error exits 0 with no output (allow), logging to
14
+ stderr (visible with --debug). This is a citation-integrity control, not a
15
+ security boundary.
16
+
17
+ URL canonicalization (borrowed from webcache): lowercase host, strip default
18
+ ports and fragments, sort query parameters — so formatting variance does not
19
+ cause silent misses.
20
+
21
+ TTL: sidecar `cached_at`; default 7 days (IUMBTEMS_FETCH_TTL_DAYS), 30 days
22
+ for stable documentation domains. Inline cap for served hits: 12000 chars
23
+ (IUMBTEMS_FETCH_MAX_INLINE), remainder via file pointer.
24
+ """
25
+
26
+ import hashlib
27
+ import json
28
+ import os
29
+ import sys
30
+ import urllib.parse
31
+ from datetime import datetime, timezone
32
+ from pathlib import Path
33
+
34
+ DEFAULT_TTL_DAYS = 7
35
+ DOC_DOMAIN_TTL_DAYS = 30
36
+ DOC_DOMAINS = (
37
+ "code.claude.com",
38
+ "docs.anthropic.com",
39
+ "platform.claude.com",
40
+ "docs.python.org",
41
+ "developer.mozilla.org",
42
+ )
43
+ DEFAULT_MAX_INLINE = 12000
44
+
45
+
46
+ def log(msg):
47
+ sys.stderr.write(f"[webcache] {msg}\n")
48
+
49
+
50
+ def canonical_url(url):
51
+ try:
52
+ p = urllib.parse.urlsplit(url.strip())
53
+ host = (p.hostname or "").lower()
54
+ if not host:
55
+ return None
56
+ port = p.port
57
+ if (p.scheme == "http" and port == 80) or (p.scheme == "https" and port == 443):
58
+ port = None
59
+ netloc = f"{host}:{port}" if port else host
60
+ query = urllib.parse.urlencode(
61
+ sorted(urllib.parse.parse_qsl(p.query, keep_blank_values=True))
62
+ )
63
+ return urllib.parse.urlunsplit((p.scheme or "https", netloc, p.path or "/", query, ""))
64
+ except Exception:
65
+ return None
66
+
67
+
68
+ def domain_of(url):
69
+ try:
70
+ return (urllib.parse.urlsplit(url).hostname or "").lower()
71
+ except Exception:
72
+ return ""
73
+
74
+
75
+ def ttl_days(url):
76
+ try:
77
+ override = int(os.environ.get("IUMBTEMS_FETCH_TTL_DAYS", ""))
78
+ return max(override, 0)
79
+ except ValueError:
80
+ pass
81
+ domain = domain_of(url)
82
+ if any(domain == d or domain.endswith("." + d) for d in DOC_DOMAINS):
83
+ return DOC_DOMAIN_TTL_DAYS
84
+ return DEFAULT_TTL_DAYS
85
+
86
+
87
+ def max_inline():
88
+ try:
89
+ return max(int(os.environ.get("IUMBTEMS_FETCH_MAX_INLINE", "")), 1000)
90
+ except ValueError:
91
+ return DEFAULT_MAX_INLINE
92
+
93
+
94
+ def research_dir(stdin_data):
95
+ base = stdin_data.get("cwd") or os.environ.get("CLAUDE_PROJECT_DIR") or os.getcwd()
96
+ return Path(base) / ".research"
97
+
98
+
99
+ def find_cached(sources_dir, url):
100
+ """Scan json sidecars for a matching canonical URL. Returns (md_path, meta) or None."""
101
+ canon = canonical_url(url)
102
+ if not canon or not sources_dir.is_dir():
103
+ return None
104
+ for sidecar in sources_dir.glob("*.json"):
105
+ try:
106
+ meta = json.loads(sidecar.read_text(encoding="utf-8"))
107
+ except (OSError, ValueError):
108
+ continue
109
+ if not isinstance(meta, dict):
110
+ continue
111
+ if canonical_url(str(meta.get("url", ""))) == canon:
112
+ md_path = sidecar.with_suffix(".md")
113
+ if md_path.exists():
114
+ return md_path, meta
115
+ return None
116
+
117
+
118
+ def is_fresh(meta, url):
119
+ try:
120
+ cached_at = datetime.fromisoformat(str(meta.get("cached_at", "")))
121
+ if cached_at.tzinfo is None:
122
+ cached_at = cached_at.replace(tzinfo=timezone.utc)
123
+ age_days = (datetime.now(timezone.utc) - cached_at).total_seconds() / 86400
124
+ return age_days <= ttl_days(url)
125
+ except (ValueError, TypeError):
126
+ return False
127
+
128
+
129
+ def cmd_gate(stdin_data):
130
+ """PreToolUse: serve fresh hits, allow everything else (fail-open)."""
131
+ tool_input = stdin_data.get("tool_input") or {}
132
+ url = tool_input.get("url", "")
133
+ if not url:
134
+ return None # allow
135
+ sources_dir = research_dir(stdin_data) / "sources"
136
+ try:
137
+ found = find_cached(sources_dir, url)
138
+ except Exception as e:
139
+ log(f"cache lookup failed for {domain_of(url)}: {e}")
140
+ return None # allow
141
+ if not found:
142
+ return None # miss -> allow, PostToolUse archives
143
+ md_path, meta = found
144
+ if not is_fresh(meta, url):
145
+ log(f"stale entry for {domain_of(url)} ({meta.get('cached_at')}); allowing live fetch")
146
+ return None # stale -> allow, PostToolUse refreshes
147
+ try:
148
+ content = md_path.read_text(encoding="utf-8")
149
+ except OSError as e:
150
+ log(f"cache read failed: {e}")
151
+ return None # allow
152
+ cap = max_inline()
153
+ served = content if len(content) <= cap else content[:cap] + "\n\n[... truncated]"
154
+ body = (
155
+ f"<!-- EPISTEMIC_CACHE_HIT: {md_path.name} -->\n"
156
+ f"**Source URL:** {url}\n"
157
+ f"**Cached:** {meta.get('cached_at')} **Content SHA-256:** `{meta.get('hash')}`\n"
158
+ f"**Verification Tag:** `[VERIFIED: {str(meta.get('hash', ''))[:16]}]`\n\n"
159
+ f"{served}\n"
160
+ )
161
+ if len(content) > cap:
162
+ body += f"\n(Full cached text in {md_path}; cite the file for passages beyond the excerpt.)\n"
163
+ log(f"cache HIT for {domain_of(url)} ({len(content)} chars, {md_path.name})")
164
+ return {
165
+ "hookSpecificOutput": {
166
+ "hookEventName": "PreToolUse",
167
+ "permissionDecision": "deny",
168
+ "permissionDecisionReason": (
169
+ "Served from the epistemic content cache (.research/sources); "
170
+ "no network fetch needed. Cite the cached file."
171
+ ),
172
+ "additionalContext": body,
173
+ }
174
+ }
175
+
176
+
177
+ def extract_response_text(stdin_data):
178
+ """Defensively locate the tool response text across field-name variants."""
179
+ for key in ("tool_response", "response", "tool_result", "result", "output"):
180
+ val = stdin_data.get(key)
181
+ if isinstance(val, str) and val.strip():
182
+ return val
183
+ if isinstance(val, dict):
184
+ for sub in ("text", "content", "output"):
185
+ inner = val.get(sub)
186
+ if isinstance(inner, str) and inner.strip():
187
+ return inner
188
+ if isinstance(inner, list):
189
+ parts = [
190
+ p.get("text", "") for p in inner
191
+ if isinstance(p, dict) and isinstance(p.get("text"), str)
192
+ ]
193
+ if parts:
194
+ return "\n".join(parts)
195
+ return ""
196
+
197
+
198
+ def extract_url(stdin_data):
199
+ tool_input = stdin_data.get("tool_input") or stdin_data.get("input") or {}
200
+ if isinstance(tool_input, dict):
201
+ url = tool_input.get("url", "")
202
+ if url:
203
+ return url
204
+ # Fall back to a URL already recorded in a response envelope.
205
+ for key in ("tool_response", "response", "tool_result"):
206
+ val = stdin_data.get(key)
207
+ if isinstance(val, dict) and val.get("url"):
208
+ return val["url"]
209
+ return ""
210
+
211
+
212
+ def cmd_archive(stdin_data):
213
+ """PostToolUse: store the live response as md + json sidecar."""
214
+ url = extract_url(stdin_data)
215
+ text = extract_response_text(stdin_data)
216
+ if not url or not text:
217
+ return None
218
+ try:
219
+ sources_dir = research_dir(stdin_data) / "sources"
220
+ sources_dir.mkdir(parents=True, exist_ok=True)
221
+ digest = hashlib.sha256(text.encode("utf-8")).hexdigest()
222
+ sources_dir.joinpath(f"{digest}.md").write_text(text, encoding="utf-8")
223
+ sources_dir.joinpath(f"{digest}.json").write_text(
224
+ json.dumps(
225
+ {
226
+ "hash": digest,
227
+ "url": url,
228
+ "title": url,
229
+ "tier": "WEB_DOCUMENT",
230
+ "cached_at": datetime.now(timezone.utc).isoformat(),
231
+ "byte_size": len(text.encode("utf-8")),
232
+ "char_count": len(text),
233
+ "custom_metadata": {"via": "webfetch-hook"},
234
+ },
235
+ indent=2,
236
+ ),
237
+ encoding="utf-8",
238
+ )
239
+ log(f"archived {domain_of(url)} -> {digest[:16]} ({len(text)} chars)")
240
+ except Exception as e:
241
+ log(f"archive failed for {domain_of(url)}: {e}")
242
+ return None
243
+
244
+
245
+ def cmd_stats(stdin_data):
246
+ """SessionStart: one-line cache summary; silent when empty."""
247
+ try:
248
+ sources_dir = research_dir(stdin_data) / "sources"
249
+ if not sources_dir.is_dir():
250
+ return None
251
+ md_files = list(sources_dir.glob("*.md"))
252
+ if not md_files:
253
+ return None
254
+ return {
255
+ "hookSpecificOutput": {
256
+ "hookEventName": "SessionStart",
257
+ "additionalContext": (
258
+ f"epistemic-cache: {len(md_files)} page(s) cached in "
259
+ f".research/sources — WebFetch hits are served from cache."
260
+ ),
261
+ }
262
+ }
263
+ except Exception as e:
264
+ log(f"stats failed: {e}")
265
+ return None
266
+
267
+
268
+ def main(argv):
269
+ mode = argv[1] if len(argv) > 1 else "gate"
270
+ try:
271
+ raw = sys.stdin.read() if not sys.stdin.isatty() else ""
272
+ stdin_data = json.loads(raw) if raw.strip() else {}
273
+ if not isinstance(stdin_data, dict):
274
+ stdin_data = {}
275
+ except (ValueError, OSError):
276
+ stdin_data = {}
277
+ handler = {"gate": cmd_gate, "archive": cmd_archive, "stats": cmd_stats}.get(mode)
278
+ if handler is None:
279
+ log(f"unknown mode: {mode}")
280
+ return 0
281
+ try:
282
+ out = handler(stdin_data)
283
+ except Exception as e:
284
+ log(f"{mode} failed: {e}")
285
+ return 0 # fail-open
286
+ if out:
287
+ sys.stdout.write(json.dumps(out))
288
+ return 0
289
+
290
+
291
+ if __name__ == "__main__":
292
+ sys.exit(main(sys.argv))