@heretek-ai/epistemic-swarm 0.7.10 → 0.7.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/package.json +1 -1
- package/plugins/research-cache/skills/research_cache/hasher.py +56 -14
- package/prompts/agent_brainstormer.md +21 -4
- package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
- package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
- package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
- package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
- package/runner/__pycache__/path_safety.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
- package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
- package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
- package/runner/__pycache__/retrieval_log.cpython-311.pyc +0 -0
- package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
- package/runner/auditor_engine.py +171 -75
- package/runner/refinement.py +74 -0
- package/runner/research_swarm.py +198 -52
- package/runner/retrieval_log.py +90 -0
- package/runner/state_machine.py +31 -13
- package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_backends.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_bet1_spike.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claude_plugin.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_dossier_contract.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_factory.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_grounding.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_manifest_concurrency.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_opencode_ux.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_opencode_v2_registrars.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_swarm_config.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_sweep_regressions.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_webcache.cpython-311.pyc +0 -0
- package/runner/tests/test_dossier_contract.py +196 -0
- package/runner/tests/test_grounding.py +183 -0
- package/scripts/__pycache__/bet1_advisory_spike.cpython-311.pyc +0 -0
- package/scripts/__pycache__/build_adapters.cpython-311.pyc +0 -0
- package/scripts/__pycache__/divergence_experiment.cpython-311.pyc +0 -0
- package/skills/epistemic_search/scripts/__pycache__/search.cpython-311.pyc +0 -0
- package/skills/epistemic_search/scripts/search.py +16 -0
- package/skills/epistemic_search/scripts/webcache.py +11 -0
- package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
- package/skills/research_cache/hasher.py +56 -14
- package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@heretek-ai/epistemic-swarm",
|
|
3
|
-
"version": "0.7.
|
|
3
|
+
"version": "0.7.12",
|
|
4
4
|
"description": "IUMBTEMS (Epistemic Swarm) — High-Integrity Dialectic Research Agent Harness for Claude Code, OpenCode V2, Pi (pi.dev), OMP (oh-my-pi), Gemini CLI, Codex CLI, and AntiGravity",
|
|
5
5
|
"main": "bin/cli.js",
|
|
6
6
|
"bin": {
|
|
@@ -17,6 +17,20 @@ from typing import Dict, Any, Optional, Tuple
|
|
|
17
17
|
DEFAULT_RESEARCH_DIR = ".research"
|
|
18
18
|
BASE_DIR_HELP = "Base .research directory"
|
|
19
19
|
|
|
20
|
+
|
|
21
|
+
def _log_retrieval(base_dir, kind: str, **fields) -> None:
|
|
22
|
+
"""Best-effort retrieval telemetry (see runner/retrieval_log.py)."""
|
|
23
|
+
try:
|
|
24
|
+
root = Path(__file__).resolve().parents[2]
|
|
25
|
+
if str(root) not in sys.path:
|
|
26
|
+
sys.path.insert(0, str(root))
|
|
27
|
+
from runner.retrieval_log import log_event
|
|
28
|
+
|
|
29
|
+
log_event(base_dir, kind, **fields)
|
|
30
|
+
except Exception:
|
|
31
|
+
pass
|
|
32
|
+
|
|
33
|
+
|
|
20
34
|
class SourceHasher:
|
|
21
35
|
def __init__(self, base_dir: Optional[Path] = None):
|
|
22
36
|
target = base_dir or Path(DEFAULT_RESEARCH_DIR)
|
|
@@ -30,8 +44,14 @@ class SourceHasher:
|
|
|
30
44
|
normalized = content.strip().encode("utf-8")
|
|
31
45
|
return hashlib.sha256(normalized).hexdigest()
|
|
32
46
|
|
|
33
|
-
def store_source(
|
|
34
|
-
|
|
47
|
+
def store_source(
|
|
48
|
+
self,
|
|
49
|
+
url: str,
|
|
50
|
+
content: str,
|
|
51
|
+
title: Optional[str] = None,
|
|
52
|
+
tier: str = "WEB_DOCUMENT",
|
|
53
|
+
metadata: Optional[Dict[str, Any]] = None,
|
|
54
|
+
) -> str:
|
|
35
55
|
"""Store content and metadata content-addressed by SHA-256."""
|
|
36
56
|
content_hash = self.compute_sha256(content)
|
|
37
57
|
md_path = self.sources_dir / f"{content_hash}.md"
|
|
@@ -50,7 +70,7 @@ class SourceHasher:
|
|
|
50
70
|
"cached_at": datetime.now(timezone.utc).isoformat(),
|
|
51
71
|
"byte_size": len(content.encode("utf-8")),
|
|
52
72
|
"char_count": len(content),
|
|
53
|
-
"custom_metadata": metadata or {}
|
|
73
|
+
"custom_metadata": metadata or {},
|
|
54
74
|
}
|
|
55
75
|
with open(json_path, "w", encoding="utf-8") as f:
|
|
56
76
|
json.dump(meta, f, indent=2)
|
|
@@ -84,7 +104,9 @@ class SourceHasher:
|
|
|
84
104
|
"""
|
|
85
105
|
if not re.match(r"^[a-fA-F0-9]+$", hash_prefix):
|
|
86
106
|
return None
|
|
87
|
-
exact_path = Path(
|
|
107
|
+
exact_path = Path(
|
|
108
|
+
os.path.realpath(str(self.sources_dir / f"{hash_prefix}{extension}"))
|
|
109
|
+
)
|
|
88
110
|
if exact_path.parent != self.sources_dir:
|
|
89
111
|
return None
|
|
90
112
|
if exact_path.exists():
|
|
@@ -102,12 +124,16 @@ class SourceHasher:
|
|
|
102
124
|
def normalize_text_for_search(text: str) -> str:
|
|
103
125
|
"""Collapse whitespace and normalize typography for substring matching."""
|
|
104
126
|
# Replace smart quotes and dashes
|
|
105
|
-
text =
|
|
127
|
+
text = (
|
|
128
|
+
text.replace("“", '"').replace("”", '"').replace("‘", "'").replace("’", "'")
|
|
129
|
+
)
|
|
106
130
|
text = text.replace("—", "-").replace("–", "-")
|
|
107
131
|
# Collapse all whitespace to single spaces
|
|
108
132
|
return re.sub(r"\s+", " ", text).strip().lower()
|
|
109
133
|
|
|
110
|
-
def verify_quote(
|
|
134
|
+
def verify_quote(
|
|
135
|
+
self, content_hash: str, quote: str
|
|
136
|
+
) -> Tuple[bool, float, Optional[str]]:
|
|
111
137
|
"""
|
|
112
138
|
Verifies whether quote exists in cached document.
|
|
113
139
|
Returns: (is_verified, confidence_score, context_match)
|
|
@@ -136,7 +162,7 @@ class SourceHasher:
|
|
|
136
162
|
best_score = 0.0
|
|
137
163
|
|
|
138
164
|
for i in range(max(1, len(source_words) - window_size + 1)):
|
|
139
|
-
window = source_words[i:i + window_size]
|
|
165
|
+
window = source_words[i : i + window_size]
|
|
140
166
|
matches = sum(1 for w1, w2 in zip(quote_words, window) if w1 == w2)
|
|
141
167
|
score = matches / window_size
|
|
142
168
|
if score > best_score:
|
|
@@ -147,11 +173,17 @@ class SourceHasher:
|
|
|
147
173
|
if best_score >= 0.88:
|
|
148
174
|
return True, best_score, f"High-confidence fuzzy match ({best_score:.2f})."
|
|
149
175
|
|
|
150
|
-
return
|
|
176
|
+
return (
|
|
177
|
+
False,
|
|
178
|
+
best_score,
|
|
179
|
+
f"Verification failed. Highest word overlap: {best_score:.2f}.",
|
|
180
|
+
)
|
|
151
181
|
|
|
152
182
|
|
|
153
183
|
def main():
|
|
154
|
-
parser = argparse.ArgumentParser(
|
|
184
|
+
parser = argparse.ArgumentParser(
|
|
185
|
+
description="Epistemic Swarm Content Hasher & Quote Verifier"
|
|
186
|
+
)
|
|
155
187
|
subparsers = parser.add_subparsers(dest="command")
|
|
156
188
|
|
|
157
189
|
# Cache command
|
|
@@ -166,7 +198,9 @@ def main():
|
|
|
166
198
|
verify_parser = subparsers.add_parser("verify", help="Verify a verbatim quote")
|
|
167
199
|
verify_parser.add_argument("--hash", required=True, help="Document SHA-256 hash")
|
|
168
200
|
verify_parser.add_argument("--quote", required=True, help="Verbatim quote to check")
|
|
169
|
-
verify_parser.add_argument(
|
|
201
|
+
verify_parser.add_argument(
|
|
202
|
+
"--dir", default=DEFAULT_RESEARCH_DIR, help=BASE_DIR_HELP
|
|
203
|
+
)
|
|
170
204
|
|
|
171
205
|
# List command
|
|
172
206
|
list_parser = subparsers.add_parser("list", help="List cached sources")
|
|
@@ -182,18 +216,26 @@ def main():
|
|
|
182
216
|
if not sys.stdin.isatty():
|
|
183
217
|
content = sys.stdin.read()
|
|
184
218
|
else:
|
|
185
|
-
print(
|
|
219
|
+
print(
|
|
220
|
+
"Error: No content provided via --content or stdin.",
|
|
221
|
+
file=sys.stderr,
|
|
222
|
+
)
|
|
186
223
|
sys.exit(1)
|
|
187
|
-
h = hasher.store_source(
|
|
224
|
+
h = hasher.store_source(
|
|
225
|
+
url=args.url, content=content, title=args.title, tier=args.tier
|
|
226
|
+
)
|
|
188
227
|
print(f"[CACHED] {h} -> {args.title} ({args.url})")
|
|
228
|
+
_log_retrieval(hasher.base_dir, "cache", url=args.url, hash=h, tool="hasher")
|
|
189
229
|
|
|
190
230
|
elif args.command == "verify":
|
|
191
|
-
verified, conf, msg = hasher.verify_quote(
|
|
231
|
+
verified, conf, msg = hasher.verify_quote(
|
|
232
|
+
content_hash=args.hash, quote=args.quote
|
|
233
|
+
)
|
|
192
234
|
result = {
|
|
193
235
|
"hash": args.hash,
|
|
194
236
|
"verified": verified,
|
|
195
237
|
"confidence": conf,
|
|
196
|
-
"message": msg
|
|
238
|
+
"message": msg,
|
|
197
239
|
}
|
|
198
240
|
print(json.dumps(result, indent=2))
|
|
199
241
|
sys.exit(0 if verified else 1)
|
|
@@ -26,6 +26,20 @@ context < 40%, use shallow mode (README + top-level tree only) and say so.
|
|
|
26
26
|
- Open loops: TODOs, failing tests, stale branches.
|
|
27
27
|
3. NEVER invent file contents. If a file was not ingested, mark the claim
|
|
28
28
|
`[HYPOTHESIS: <how to check>]`, never `[VERIFIED]`.
|
|
29
|
+
4. **External grounding** (mandatory when the objective implies external facts,
|
|
30
|
+
benchmarks, or prior art): discover with the zero-key search skill and cache
|
|
31
|
+
every source before citing it:
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
python3 skills/epistemic_search/scripts/search.py "<query>"
|
|
35
|
+
python3 skills/research_cache/hasher.py cache --url "<URL>" --content "$(cat fetched.md)" --title "<TITLE>"
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
Cite a cached source as `[VERIFIED: <sha256>]` **with an exact
|
|
39
|
+
`verbatim_quote`**; anything not retrieved stays `[HYPOTHESIS: <test>]`. A run
|
|
40
|
+
that never searches yields zero cached sources and will be flagged
|
|
41
|
+
`WARNING_LOW_GROUNDING` — that is only acceptable when the objective is purely
|
|
42
|
+
internal to the workspace.
|
|
29
43
|
|
|
30
44
|
## 2. DIALECTICAL DIVERGENCE (mandatory, in order)
|
|
31
45
|
|
|
@@ -81,10 +95,11 @@ Write `.research/brainstorm_<slug>.md`:
|
|
|
81
95
|
<what you deliberately did NOT propose: bug fixes, chores, refactors>
|
|
82
96
|
```
|
|
83
97
|
|
|
84
|
-
Also write
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
98
|
+
Also write your **role dossier** as JSON, to the exact path the runner names in
|
|
99
|
+
your task (Agent Alpha → `alpha_dossier.json`, Agent Beta → `beta_dossier.json`).
|
|
100
|
+
It must be dossier-shaped and `scope_id`-keyed; claims use `HYPOTHESIS` with a
|
|
101
|
+
`falsification` field, and `VERIFIED` only for ingested workspace facts with
|
|
102
|
+
`file://` pointers plus verbatim snippets.
|
|
88
103
|
|
|
89
104
|
## 4. HARD BANS
|
|
90
105
|
|
|
@@ -95,3 +110,5 @@ workspace facts with `file://` pointers and verbatim snippets).
|
|
|
95
110
|
`[HYPOTHESIS]` or `[INFERRED: <parents>]`.
|
|
96
111
|
4. No writes outside `.research/`.
|
|
97
112
|
5. No scope creep past 2 dialectic iterations without user confirmation.
|
|
113
|
+
6. Do NOT create or modify `manifest.json` in the scratchpad — it is
|
|
114
|
+
runner-owned. Emit only your role dossier (and the mode's `.md` artifact).
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
package/runner/auditor_engine.py
CHANGED
|
@@ -17,6 +17,22 @@ from skills.research_cache.hasher import SourceHasher
|
|
|
17
17
|
from runner.state_machine import ResearchStateMachine, ScopeStatus
|
|
18
18
|
from runner.refinement import compute_epistemic_score
|
|
19
19
|
|
|
20
|
+
|
|
21
|
+
def _mark_rejected_claims(score_claims: list, *result_lists: list) -> None:
|
|
22
|
+
"""Mirror the auditor's UNVERIFIED_REJECTED verdict onto ClaimWitness records."""
|
|
23
|
+
from runner.claim_witness import STATUS_REJECTED
|
|
24
|
+
|
|
25
|
+
results_by_id: Dict[str, Any] = {}
|
|
26
|
+
for results in result_lists:
|
|
27
|
+
for r in results:
|
|
28
|
+
results_by_id[r["claim_id"]] = r
|
|
29
|
+
for c in score_claims:
|
|
30
|
+
r = results_by_id.get(c.claim_id)
|
|
31
|
+
if r is not None and r.get("audited_tag") == "UNVERIFIED_REJECTED":
|
|
32
|
+
c.tag = "UNVERIFIED_REJECTED"
|
|
33
|
+
c.status = STATUS_REJECTED
|
|
34
|
+
|
|
35
|
+
|
|
20
36
|
def _compute_scope_epistemic_score(
|
|
21
37
|
constitution: Any,
|
|
22
38
|
alpha_dossier: Dict[str, Any],
|
|
@@ -24,20 +40,32 @@ def _compute_scope_epistemic_score(
|
|
|
24
40
|
alpha_results: list,
|
|
25
41
|
beta_results: list,
|
|
26
42
|
counts: Dict[str, int],
|
|
43
|
+
mode: Optional[str] = None,
|
|
27
44
|
) -> float:
|
|
28
45
|
from runner.refinement import LEGACY_CONSTITUTION
|
|
29
46
|
|
|
47
|
+
if mode == "brainstorm":
|
|
48
|
+
# Lateral ideation is scored on well-formed speculation, not web quotes.
|
|
49
|
+
from runner.claim_witness import claims_from_dossier
|
|
50
|
+
from runner.refinement import compute_brainstorm_score_from_claims
|
|
51
|
+
|
|
52
|
+
score_claims = claims_from_dossier(alpha_dossier) + claims_from_dossier(
|
|
53
|
+
beta_dossier
|
|
54
|
+
)
|
|
55
|
+
_mark_rejected_claims(score_claims, alpha_results, beta_results)
|
|
56
|
+
epistemic_score, _breakdown = compute_brainstorm_score_from_claims(
|
|
57
|
+
score_claims, constitution=constitution
|
|
58
|
+
)
|
|
59
|
+
return epistemic_score
|
|
60
|
+
|
|
30
61
|
if constitution is not LEGACY_CONSTITUTION:
|
|
31
|
-
from runner.claim_witness import
|
|
62
|
+
from runner.claim_witness import claims_from_dossier
|
|
32
63
|
from runner.refinement import compute_epistemic_score_from_claims
|
|
33
64
|
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
if r is not None and r.get("audited_tag") == "UNVERIFIED_REJECTED":
|
|
39
|
-
c.tag = "UNVERIFIED_REJECTED"
|
|
40
|
-
c.status = STATUS_REJECTED
|
|
65
|
+
score_claims = claims_from_dossier(alpha_dossier) + claims_from_dossier(
|
|
66
|
+
beta_dossier
|
|
67
|
+
)
|
|
68
|
+
_mark_rejected_claims(score_claims, alpha_results, beta_results)
|
|
41
69
|
epistemic_score, _breakdown = compute_epistemic_score_from_claims(
|
|
42
70
|
score_claims, constitution=constitution
|
|
43
71
|
)
|
|
@@ -65,11 +93,17 @@ def _apply_living_dossiers_degradation(
|
|
|
65
93
|
|
|
66
94
|
retractions = load_retractions(base_dir)
|
|
67
95
|
if retractions:
|
|
68
|
-
all_claims = claims_from_dossier(alpha_dossier) + claims_from_dossier(
|
|
69
|
-
|
|
96
|
+
all_claims = claims_from_dossier(alpha_dossier) + claims_from_dossier(
|
|
97
|
+
beta_dossier
|
|
98
|
+
)
|
|
99
|
+
_degraded, events = apply_degradation(
|
|
100
|
+
all_claims, retractions, scope_id=scope_id
|
|
101
|
+
)
|
|
70
102
|
if events:
|
|
71
103
|
append_ledger(base_dir, events)
|
|
72
|
-
queue_requeue(
|
|
104
|
+
queue_requeue(
|
|
105
|
+
base_dir, scope_id, reason="claim degradation (retraction event)"
|
|
106
|
+
)
|
|
73
107
|
degradation = {
|
|
74
108
|
"events": [e.to_dict() for e in events],
|
|
75
109
|
"degraded_scopes": [scope_id] if events else [],
|
|
@@ -85,14 +119,20 @@ class EpistemicAuditorEngine:
|
|
|
85
119
|
self.hasher = SourceHasher(base_dir=self.base_dir)
|
|
86
120
|
self.state_machine = ResearchStateMachine(base_dir=self.base_dir)
|
|
87
121
|
|
|
88
|
-
def audit_scope(
|
|
122
|
+
def audit_scope(
|
|
123
|
+
self,
|
|
124
|
+
scope_id: str,
|
|
125
|
+
constitution: Optional[Any] = None,
|
|
126
|
+
mode: Optional[str] = None,
|
|
127
|
+
) -> Dict[str, Any]:
|
|
89
128
|
"""Runs the audit pipeline on a scope with ready dossiers.
|
|
90
129
|
|
|
91
130
|
`constitution` (Stream G) is an optional Domain Pack. Omitting it
|
|
92
131
|
preserves legacy behavior exactly (flat 1.0 weights, 0.65 threshold,
|
|
93
132
|
no domain/tag rules). Passing one enables tier-weighted scoring and
|
|
94
133
|
the pack's per-claim gates (banned domains, mandatory tags,
|
|
95
|
-
retraction policy).
|
|
134
|
+
retraction policy). `mode` selects the scoring contract — brainstorm
|
|
135
|
+
credits well-formed speculation rather than web-quote grounding.
|
|
96
136
|
"""
|
|
97
137
|
from runner.refinement import LEGACY_CONSTITUTION
|
|
98
138
|
|
|
@@ -126,14 +166,25 @@ class EpistemicAuditorEngine:
|
|
|
126
166
|
"rejected": alpha_rejected + beta_rejected,
|
|
127
167
|
"inferred": len(alpha_dossier.get("inferred_implications", [])),
|
|
128
168
|
"hypotheses": len(beta_dossier.get("hypotheses", [])),
|
|
129
|
-
"neg_knowledge": len(alpha_dossier.get("negative_knowledge", []))
|
|
169
|
+
"neg_knowledge": len(alpha_dossier.get("negative_knowledge", []))
|
|
170
|
+
+ len(beta_dossier.get("negative_knowledge", [])),
|
|
130
171
|
}
|
|
131
172
|
|
|
132
173
|
# 3. Calculate Epistemic Score via the pure refinement function.
|
|
133
174
|
epistemic_score = _compute_scope_epistemic_score(
|
|
134
|
-
constitution,
|
|
175
|
+
constitution,
|
|
176
|
+
alpha_dossier,
|
|
177
|
+
beta_dossier,
|
|
178
|
+
alpha_results,
|
|
179
|
+
beta_results,
|
|
180
|
+
counts,
|
|
181
|
+
mode=mode,
|
|
135
182
|
)
|
|
136
183
|
accept_threshold = constitution.accept_threshold
|
|
184
|
+
if mode == "brainstorm":
|
|
185
|
+
from runner.refinement import BRAINSTORM_MODE_THRESHOLD
|
|
186
|
+
|
|
187
|
+
accept_threshold = min(accept_threshold, BRAINSTORM_MODE_THRESHOLD)
|
|
137
188
|
|
|
138
189
|
# 3b. Living Dossiers (Stream C): check degradation against retraction events
|
|
139
190
|
degradation = _apply_living_dossiers_degradation(
|
|
@@ -141,7 +192,9 @@ class EpistemicAuditorEngine:
|
|
|
141
192
|
)
|
|
142
193
|
|
|
143
194
|
# 4. Calculate Divergence Score
|
|
144
|
-
divergence_score, divergence_matrix = self._compute_divergence(
|
|
195
|
+
divergence_score, divergence_matrix = self._compute_divergence(
|
|
196
|
+
alpha_dossier, beta_dossier
|
|
197
|
+
)
|
|
145
198
|
|
|
146
199
|
# 5. Build Audit Report
|
|
147
200
|
audit_report = {
|
|
@@ -155,12 +208,15 @@ class EpistemicAuditorEngine:
|
|
|
155
208
|
"negative_knowledge_count": counts["neg_knowledge"],
|
|
156
209
|
"epistemic_score": epistemic_score,
|
|
157
210
|
"divergence_score": divergence_score,
|
|
158
|
-
"
|
|
211
|
+
"mode": mode,
|
|
212
|
+
"verdict": "CERTIFIED"
|
|
213
|
+
if epistemic_score >= accept_threshold
|
|
214
|
+
else "WARNING_LOW_GROUNDING",
|
|
159
215
|
},
|
|
160
216
|
"alpha_claims_audit": alpha_results,
|
|
161
217
|
"beta_claims_audit": beta_results,
|
|
162
218
|
"divergence_matrix": divergence_matrix,
|
|
163
|
-
"degradation": degradation
|
|
219
|
+
"degradation": degradation,
|
|
164
220
|
}
|
|
165
221
|
|
|
166
222
|
# Save audit_report.json
|
|
@@ -219,14 +275,16 @@ class EpistemicAuditorEngine:
|
|
|
219
275
|
statement = claim.get("statement", "")
|
|
220
276
|
|
|
221
277
|
if not shash or not quote:
|
|
222
|
-
audited_claims.append(
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
278
|
+
audited_claims.append(
|
|
279
|
+
{
|
|
280
|
+
"claim_id": cid,
|
|
281
|
+
"statement": statement,
|
|
282
|
+
"original_tag": claim.get("tag", "VERIFIED"),
|
|
283
|
+
"audited_tag": "UNVERIFIED_REJECTED",
|
|
284
|
+
"reason": "Missing source_hash or verbatim_quote",
|
|
285
|
+
"confidence": 0.0,
|
|
286
|
+
}
|
|
287
|
+
)
|
|
230
288
|
rejected_count += 1
|
|
231
289
|
continue
|
|
232
290
|
|
|
@@ -235,47 +293,54 @@ class EpistemicAuditorEngine:
|
|
|
235
293
|
if passed and claim_gate is not None:
|
|
236
294
|
verdict = claim_gate(claim)
|
|
237
295
|
if verdict["verdict"] == "REJECTED":
|
|
238
|
-
audited_claims.append(
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
296
|
+
audited_claims.append(
|
|
297
|
+
{
|
|
298
|
+
"claim_id": cid,
|
|
299
|
+
"statement": statement,
|
|
300
|
+
"original_tag": claim.get("tag", "VERIFIED"),
|
|
301
|
+
"audited_tag": "UNVERIFIED_REJECTED",
|
|
302
|
+
"source_hash": shash,
|
|
303
|
+
"rejected_quote": quote,
|
|
304
|
+
"reason": "; ".join(verdict["reasons"]),
|
|
305
|
+
"confidence": conf,
|
|
306
|
+
}
|
|
307
|
+
)
|
|
248
308
|
rejected_count += 1
|
|
249
309
|
continue
|
|
250
310
|
|
|
251
311
|
if passed:
|
|
252
|
-
audited_claims.append(
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
312
|
+
audited_claims.append(
|
|
313
|
+
{
|
|
314
|
+
"claim_id": cid,
|
|
315
|
+
"statement": statement,
|
|
316
|
+
"original_tag": "VERIFIED",
|
|
317
|
+
"audited_tag": "VERIFIED",
|
|
318
|
+
"source_hash": shash,
|
|
319
|
+
"confidence": conf,
|
|
320
|
+
"verification_message": msg,
|
|
321
|
+
}
|
|
322
|
+
)
|
|
261
323
|
verified_count += 1
|
|
262
324
|
else:
|
|
263
|
-
audited_claims.append(
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
325
|
+
audited_claims.append(
|
|
326
|
+
{
|
|
327
|
+
"claim_id": cid,
|
|
328
|
+
"statement": statement,
|
|
329
|
+
"original_tag": "VERIFIED",
|
|
330
|
+
"audited_tag": "UNVERIFIED_REJECTED",
|
|
331
|
+
"source_hash": shash,
|
|
332
|
+
"rejected_quote": quote,
|
|
333
|
+
"reason": msg,
|
|
334
|
+
"confidence": conf,
|
|
335
|
+
}
|
|
336
|
+
)
|
|
273
337
|
rejected_count += 1
|
|
274
338
|
|
|
275
339
|
return audited_claims, verified_count, rejected_count
|
|
276
340
|
|
|
277
|
-
def _compute_divergence(
|
|
278
|
-
|
|
341
|
+
def _compute_divergence(
|
|
342
|
+
self, alpha_dossier: Dict[str, Any], beta_dossier: Dict[str, Any]
|
|
343
|
+
) -> Tuple[float, List[Dict[str, Any]]]:
|
|
279
344
|
alpha_claims = alpha_dossier.get("affirmative_claims", [])
|
|
280
345
|
beta_claims = beta_dossier.get("falsification_claims", [])
|
|
281
346
|
critiques = beta_dossier.get("methodological_critiques", [])
|
|
@@ -285,12 +350,14 @@ class EpistemicAuditorEngine:
|
|
|
285
350
|
|
|
286
351
|
for crit in critiques:
|
|
287
352
|
target = crit.get("target_assertion", "")
|
|
288
|
-
matrix.append(
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
353
|
+
matrix.append(
|
|
354
|
+
{
|
|
355
|
+
"tension_type": "METHODOLOGICAL_CHALLENGE",
|
|
356
|
+
"proponent_claim": target,
|
|
357
|
+
"adversary_critique": crit.get("critique", ""),
|
|
358
|
+
"counter_evidence_hash": crit.get("evidence_hash", "NONE"),
|
|
359
|
+
}
|
|
360
|
+
)
|
|
294
361
|
contradictions += 1
|
|
295
362
|
|
|
296
363
|
# Calculate divergence
|
|
@@ -303,10 +370,14 @@ class EpistemicAuditorEngine:
|
|
|
303
370
|
lines = ["## 1. Verified Empirical Grounding"]
|
|
304
371
|
for claim in alpha_audit:
|
|
305
372
|
if claim["audited_tag"] == "VERIFIED":
|
|
306
|
-
lines.append(
|
|
373
|
+
lines.append(
|
|
374
|
+
f"- `[VERIFIED: {claim['source_hash'][:8]}]` {claim['statement']}"
|
|
375
|
+
)
|
|
307
376
|
for claim in beta_audit:
|
|
308
377
|
if claim["audited_tag"] == "VERIFIED":
|
|
309
|
-
lines.append(
|
|
378
|
+
lines.append(
|
|
379
|
+
f"- `[VERIFIED: {claim['source_hash'][:8]}]` (Counter-Evidence) {claim['statement']}"
|
|
380
|
+
)
|
|
310
381
|
return lines
|
|
311
382
|
|
|
312
383
|
@staticmethod
|
|
@@ -316,9 +387,13 @@ class EpistemicAuditorEngine:
|
|
|
316
387
|
for item in matrix:
|
|
317
388
|
lines.append(f"### Tension: {item['tension_type']}")
|
|
318
389
|
lines.append(f"- **Thesis Assertion**: {item['proponent_claim']}")
|
|
319
|
-
lines.append(
|
|
390
|
+
lines.append(
|
|
391
|
+
f"- **Adversarial Critique**: {item['adversary_critique']}"
|
|
392
|
+
)
|
|
320
393
|
else:
|
|
321
|
-
lines.append(
|
|
394
|
+
lines.append(
|
|
395
|
+
"No active contradictions identified between primary literature and red-team findings."
|
|
396
|
+
)
|
|
322
397
|
return lines
|
|
323
398
|
|
|
324
399
|
@staticmethod
|
|
@@ -327,15 +402,23 @@ class EpistemicAuditorEngine:
|
|
|
327
402
|
rejected = [c for c in claims if c["audited_tag"] == "UNVERIFIED_REJECTED"]
|
|
328
403
|
if rejected:
|
|
329
404
|
for r in rejected:
|
|
330
|
-
lines.append(
|
|
405
|
+
lines.append(
|
|
406
|
+
f'- ⚠️ **PURGED**: "{r["statement"]}" — *Reason: {r["reason"]}*'
|
|
407
|
+
)
|
|
331
408
|
else:
|
|
332
|
-
lines.append(
|
|
409
|
+
lines.append(
|
|
410
|
+
"Zero claims rejected. 100% of cited assertions verified against source cache."
|
|
411
|
+
)
|
|
333
412
|
return lines
|
|
334
413
|
|
|
335
414
|
@staticmethod
|
|
336
|
-
def _render_negative_knowledge_section(
|
|
415
|
+
def _render_negative_knowledge_section(
|
|
416
|
+
alpha: Dict[str, Any], beta: Dict[str, Any]
|
|
417
|
+
) -> List[str]:
|
|
337
418
|
lines = ["\n## 4. Negative Knowledge Catalog"]
|
|
338
|
-
all_neg = alpha.get("negative_knowledge", []) + beta.get(
|
|
419
|
+
all_neg = alpha.get("negative_knowledge", []) + beta.get(
|
|
420
|
+
"negative_knowledge", []
|
|
421
|
+
)
|
|
339
422
|
if all_neg:
|
|
340
423
|
for n in all_neg:
|
|
341
424
|
lines.append(f"- `[NEGATIVE_KNOWLEDGE: {n['query']}]` {n['finding']}")
|
|
@@ -343,15 +426,28 @@ class EpistemicAuditorEngine:
|
|
|
343
426
|
lines.append("No negative knowledge declarations logged.")
|
|
344
427
|
return lines
|
|
345
428
|
|
|
346
|
-
def _generate_synthesis_markdown(
|
|
347
|
-
|
|
429
|
+
def _generate_synthesis_markdown(
|
|
430
|
+
self,
|
|
431
|
+
scope_id: str,
|
|
432
|
+
alpha: Dict[str, Any],
|
|
433
|
+
beta: Dict[str, Any],
|
|
434
|
+
audit: Dict[str, Any],
|
|
435
|
+
) -> str:
|
|
348
436
|
summary = audit["summary"]
|
|
349
437
|
md = [
|
|
350
438
|
f"# Epistemic Synthesis: Scope {scope_id}\n",
|
|
351
439
|
f"**Audit Verdict**: `{summary['verdict']}` | **Epistemic Score**: `{summary['epistemic_score']}/1.0` | **Divergence Index**: `{summary['divergence_score']}`\n",
|
|
352
440
|
]
|
|
353
|
-
md.extend(
|
|
441
|
+
md.extend(
|
|
442
|
+
self._render_verified_section(
|
|
443
|
+
audit["alpha_claims_audit"], audit["beta_claims_audit"]
|
|
444
|
+
)
|
|
445
|
+
)
|
|
354
446
|
md.extend(self._render_tensions_section(audit.get("divergence_matrix", [])))
|
|
355
|
-
md.extend(
|
|
447
|
+
md.extend(
|
|
448
|
+
self._render_rejected_section(
|
|
449
|
+
audit["alpha_claims_audit"] + audit["beta_claims_audit"]
|
|
450
|
+
)
|
|
451
|
+
)
|
|
356
452
|
md.extend(self._render_negative_knowledge_section(alpha, beta))
|
|
357
453
|
return "\n".join(md)
|