@heretek-ai/epistemic-swarm 0.2.2 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/skills/brainstorming/SKILL.md +13 -0
- package/.agents/skills/code_audit/SKILL.md +13 -0
- package/.agents/skills/epistemic_search/SKILL.md +13 -0
- package/.agents/skills/grilling/SKILL.md +13 -0
- package/.agents/skills/oss_scout/SKILL.md +13 -0
- package/.agents/skills/research_cache/SKILL.md +13 -0
- package/.agents/skills/swarm_config/SKILL.md +13 -0
- package/.claude-plugin/plugin.json +15 -5
- package/.omp/README.md +39 -0
- package/.omp/SYSTEM.md +12 -0
- package/.omp/commands/audit.md +10 -0
- package/.omp/commands/brainstorming.md +12 -0
- package/.omp/commands/grill.md +9 -0
- package/.omp/commands/scout.md +10 -0
- package/.omp/commands/swarm-config.md +10 -0
- package/.omp/commands/swarm.md +9 -0
- package/.omp/hooks/post/epistemic-audit.ts +16 -0
- package/.omp/hooks/pre/epistemic-redirect.ts +21 -0
- package/.omp/prompts/brainstorming.md +8 -0
- package/.omp/prompts/swarm.md +8 -0
- package/MARKETPLACE.md +8 -0
- package/README.md +53 -8
- package/bin/cli.js +27 -3
- package/config/domain_packs/biopharma.json +23 -0
- package/config/domain_packs/legal.json +19 -0
- package/config/domain_packs/quant.json +19 -0
- package/config/mcp-research-servers.json +7 -0
- package/config/mcp_launcher.py +48 -137
- package/config/opencode-snippet.json +58 -3
- package/config/searxng_mcp.py +42 -83
- package/extensions/pi/index.js +196 -28
- package/install.sh +20 -4
- package/package.json +40 -5
- package/plugins/antigravity/README.md +28 -0
- package/plugins/antigravity/agents/alpha-thesis.md +6 -0
- package/plugins/antigravity/agents/beta-antithesis.md +7 -0
- package/plugins/antigravity/agents/brainstormer.md +7 -0
- package/plugins/antigravity/agents/epistemic-auditor.md +5 -0
- package/plugins/antigravity/hooks.json +23 -0
- package/plugins/antigravity/mcp_config.json +33 -0
- package/plugins/antigravity/plugin.json +21 -0
- package/plugins/antigravity/rules/epistemic-integrity.md +6 -0
- package/plugins/antigravity/skills/brainstorming/SKILL.md +13 -0
- package/plugins/antigravity/skills/code_audit/SKILL.md +13 -0
- package/plugins/antigravity/skills/epistemic_search/SKILL.md +13 -0
- package/plugins/antigravity/skills/grilling/SKILL.md +13 -0
- package/plugins/antigravity/skills/oss_scout/SKILL.md +13 -0
- package/plugins/antigravity/skills/research_cache/SKILL.md +13 -0
- package/plugins/antigravity/skills/swarm_config/SKILL.md +13 -0
- package/plugins/codex/AGENTS.md.snippet +10 -0
- package/plugins/codex/README.md +37 -0
- package/plugins/codex/config.toml.snippet +28 -0
- package/plugins/codex/openai.yaml +24 -0
- package/plugins/codex/skills/brainstorming/SKILL.md +13 -0
- package/plugins/codex/skills/code_audit/SKILL.md +13 -0
- package/plugins/codex/skills/epistemic_search/SKILL.md +13 -0
- package/plugins/codex/skills/grilling/SKILL.md +13 -0
- package/plugins/codex/skills/oss_scout/SKILL.md +13 -0
- package/plugins/codex/skills/research_cache/SKILL.md +13 -0
- package/plugins/codex/skills/swarm_config/SKILL.md +13 -0
- package/plugins/gemini/GEMINI.md +15 -0
- package/plugins/gemini/README.md +19 -0
- package/plugins/gemini/commands/audit.toml +6 -0
- package/plugins/gemini/commands/brainstorming.toml +10 -0
- package/plugins/gemini/commands/grill.toml +6 -0
- package/plugins/gemini/commands/scout.toml +7 -0
- package/plugins/gemini/commands/swarm-config.toml +7 -0
- package/plugins/gemini/commands/swarm.toml +8 -0
- package/plugins/gemini/gemini-extension.json +38 -0
- package/plugins/gemini/hooks/hooks.json +11 -0
- package/plugins/gemini/skills/brainstorming/SKILL.md +13 -0
- package/plugins/gemini/skills/code_audit/SKILL.md +13 -0
- package/plugins/gemini/skills/epistemic_search/SKILL.md +13 -0
- package/plugins/gemini/skills/grilling/SKILL.md +13 -0
- package/plugins/gemini/skills/oss_scout/SKILL.md +13 -0
- package/plugins/gemini/skills/research_cache/SKILL.md +13 -0
- package/plugins/gemini/skills/swarm_config/SKILL.md +13 -0
- package/plugins/opencode/index.js +335 -118
- package/prompts/agent_brainstormer.md +97 -0
- package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
- package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
- package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
- package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
- package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
- package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
- package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
- package/runner/auctioneer.py +169 -0
- package/runner/auditor_engine.py +127 -12
- package/runner/claim_store.py +361 -0
- package/runner/claim_witness.py +183 -0
- package/runner/living_dossiers.py +355 -0
- package/runner/mcp_protocol.py +188 -0
- package/runner/mcp_server.py +567 -0
- package/runner/pcrb.py +212 -0
- package/runner/pcrb_verify.py +237 -0
- package/runner/refinement.py +335 -0
- package/runner/research_swarm.py +378 -94
- package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
- package/runner/tests/fixtures/auction_objective.json +35 -0
- package/runner/tests/fixtures/borderline_claims.json +27 -0
- package/runner/tests/fixtures/divergence_objectives.json +31 -0
- package/runner/tests/test_auction_order.py +173 -0
- package/runner/tests/test_claim_store.py +162 -0
- package/runner/tests/test_claim_witness.py +186 -0
- package/runner/tests/test_domain_packs.py +217 -0
- package/runner/tests/test_fleet_seam.py +113 -0
- package/runner/tests/test_living_dossiers.py +204 -0
- package/runner/tests/test_mcp_server.py +212 -0
- package/runner/tests/test_pcrb.py +240 -0
- package/runner/tests/test_refinement.py +255 -0
- package/runner/tests/test_swarm.py +173 -15
- package/scripts/auction_experiment.py +180 -0
- package/scripts/build_adapters.py +183 -0
- package/scripts/divergence_experiment.py +184 -0
- package/skills/brainstorming/SKILL.md +106 -0
- package/skills/brainstorming/__init__.py +1 -0
- package/skills/brainstorming/scripts/brainstorm.py +200 -0
- package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
- package/skills/swarm_config/SKILL.md +1 -1
- package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
- package/skills/swarm_config/configure.py +45 -11
|
@@ -0,0 +1,361 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Derived claim index over the .research/ flat-file record.
|
|
4
|
+
|
|
5
|
+
INVIOLABLE: .research/**/*.json|md stays the source of truth. This sqlite
|
|
6
|
+
DB is a rebuildable index for cross-session dedup, degradation queries, and
|
|
7
|
+
provenance joins. If it ever drifts: delete it and `--reindex`. Never
|
|
8
|
+
"fix the DB" in place.
|
|
9
|
+
|
|
10
|
+
Tables:
|
|
11
|
+
claims one row per ClaimWitness
|
|
12
|
+
sources one row per content hash (from .research/sources/<hash>.json)
|
|
13
|
+
claim_sources join (claim_id -> source_hash)
|
|
14
|
+
status_events append-only degradation log (Stream C mirrors into here)
|
|
15
|
+
|
|
16
|
+
Zero external deps: stdlib sqlite3. FTS5 over `statement` when compiled in,
|
|
17
|
+
else a LIKE scan. An `embedding BLOB` column is reserved for the vector half
|
|
18
|
+
of the dual-layer design; embeddings are deliberately NOT built here (they
|
|
19
|
+
need an API call and are out of this phase's scope).
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
import argparse
|
|
23
|
+
import hashlib
|
|
24
|
+
import json
|
|
25
|
+
import sqlite3
|
|
26
|
+
import sys
|
|
27
|
+
from pathlib import Path
|
|
28
|
+
from typing import Any, Dict, Iterable, List, Optional, Tuple
|
|
29
|
+
|
|
30
|
+
SCRIPT_DIR = Path(__file__).resolve().parent
|
|
31
|
+
PROJECT_ROOT = SCRIPT_DIR.parent
|
|
32
|
+
if str(PROJECT_ROOT) not in sys.path:
|
|
33
|
+
sys.path.insert(0, str(PROJECT_ROOT))
|
|
34
|
+
|
|
35
|
+
from runner.claim_witness import ClaimWitness, claims_from_dossier # noqa: E402
|
|
36
|
+
|
|
37
|
+
SCHEMA_VERSION = 1
|
|
38
|
+
|
|
39
|
+
DOSIER_NAMES = ("alpha_dossier.json", "beta_dossier.json")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _has_fts5(conn: sqlite3.Connection) -> bool:
|
|
43
|
+
try:
|
|
44
|
+
conn.execute("CREATE VIRTUAL TABLE IF NOT EXISTS _fts5_probe USING fts5(x)")
|
|
45
|
+
conn.execute("DROP TABLE _fts5_probe")
|
|
46
|
+
return True
|
|
47
|
+
except sqlite3.OperationalError:
|
|
48
|
+
return False
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class ClaimStore:
|
|
52
|
+
def __init__(self, base_dir: Path, db_path: Optional[Path] = None):
|
|
53
|
+
self.base_dir = Path(base_dir)
|
|
54
|
+
self.db_path = db_path or (self.base_dir / "claims.sqlite")
|
|
55
|
+
self.db_path.parent.mkdir(parents=True, exist_ok=True)
|
|
56
|
+
self.conn = sqlite3.connect(str(self.db_path))
|
|
57
|
+
self.conn.row_factory = sqlite3.Row
|
|
58
|
+
self.fts5 = _has_fts5(self.conn)
|
|
59
|
+
self._create_schema()
|
|
60
|
+
|
|
61
|
+
def _create_schema(self) -> None:
|
|
62
|
+
cur = self.conn.cursor()
|
|
63
|
+
cur.executescript(
|
|
64
|
+
"""
|
|
65
|
+
CREATE TABLE IF NOT EXISTS meta (
|
|
66
|
+
key TEXT PRIMARY KEY,
|
|
67
|
+
value TEXT NOT NULL
|
|
68
|
+
);
|
|
69
|
+
CREATE TABLE IF NOT EXISTS sources (
|
|
70
|
+
source_hash TEXT PRIMARY KEY,
|
|
71
|
+
url TEXT,
|
|
72
|
+
title TEXT,
|
|
73
|
+
tier TEXT,
|
|
74
|
+
byte_size INTEGER,
|
|
75
|
+
char_count INTEGER
|
|
76
|
+
);
|
|
77
|
+
CREATE TABLE IF NOT EXISTS claims (
|
|
78
|
+
claim_id TEXT NOT NULL,
|
|
79
|
+
scope_id TEXT,
|
|
80
|
+
dossier TEXT,
|
|
81
|
+
kind TEXT,
|
|
82
|
+
tag TEXT,
|
|
83
|
+
statement TEXT,
|
|
84
|
+
source_hash TEXT,
|
|
85
|
+
source_url TEXT,
|
|
86
|
+
verbatim_quote TEXT,
|
|
87
|
+
parent_claims TEXT,
|
|
88
|
+
deductive_logic TEXT,
|
|
89
|
+
falsification TEXT,
|
|
90
|
+
query TEXT,
|
|
91
|
+
finding TEXT,
|
|
92
|
+
severity TEXT,
|
|
93
|
+
tier TEXT,
|
|
94
|
+
confidence REAL,
|
|
95
|
+
status TEXT,
|
|
96
|
+
embedding BLOB,
|
|
97
|
+
PRIMARY KEY (scope_id, dossier, claim_id)
|
|
98
|
+
);
|
|
99
|
+
CREATE TABLE IF NOT EXISTS claim_sources (
|
|
100
|
+
scope_id TEXT,
|
|
101
|
+
dossier TEXT,
|
|
102
|
+
claim_id TEXT,
|
|
103
|
+
source_hash TEXT,
|
|
104
|
+
PRIMARY KEY (scope_id, dossier, claim_id, source_hash)
|
|
105
|
+
);
|
|
106
|
+
CREATE TABLE IF NOT EXISTS status_events (
|
|
107
|
+
event_id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
108
|
+
scope_id TEXT,
|
|
109
|
+
claim_id TEXT,
|
|
110
|
+
from_status TEXT,
|
|
111
|
+
to_status TEXT,
|
|
112
|
+
reason TEXT,
|
|
113
|
+
at TEXT
|
|
114
|
+
);
|
|
115
|
+
"""
|
|
116
|
+
)
|
|
117
|
+
cur.execute("INSERT OR IGNORE INTO meta(key, value) VALUES ('schema_version', ?)", (str(SCHEMA_VERSION),))
|
|
118
|
+
if self.fts5:
|
|
119
|
+
cur.execute(
|
|
120
|
+
"""
|
|
121
|
+
CREATE VIRTUAL TABLE IF NOT EXISTS claims_fts
|
|
122
|
+
USING fts5(claim_id, statement, content='claims', content_rowid='rowid')
|
|
123
|
+
"""
|
|
124
|
+
)
|
|
125
|
+
self.conn.commit()
|
|
126
|
+
|
|
127
|
+
# ---- indexing -------------------------------------------------------
|
|
128
|
+
|
|
129
|
+
def index_dossier(
|
|
130
|
+
self, scope_id: str, dossier_name: str, dossier: Dict[str, Any]
|
|
131
|
+
) -> int:
|
|
132
|
+
"""Index one dossier dict. Idempotent: re-indexing replaces rows."""
|
|
133
|
+
claims = claims_from_dossier(dossier)
|
|
134
|
+
cur = self.conn.cursor()
|
|
135
|
+
cur.execute(
|
|
136
|
+
"DELETE FROM claims WHERE scope_id = ? AND dossier = ?",
|
|
137
|
+
(scope_id, dossier_name),
|
|
138
|
+
)
|
|
139
|
+
cur.execute(
|
|
140
|
+
"DELETE FROM claim_sources WHERE scope_id = ? AND dossier = ?",
|
|
141
|
+
(scope_id, dossier_name),
|
|
142
|
+
)
|
|
143
|
+
for c in claims:
|
|
144
|
+
self._upsert_claim(cur, scope_id, dossier_name, c)
|
|
145
|
+
self.conn.commit()
|
|
146
|
+
self._sync_fts()
|
|
147
|
+
return len(claims)
|
|
148
|
+
|
|
149
|
+
def _upsert_claim(
|
|
150
|
+
self, cur: sqlite3.Cursor, scope_id: str, dossier: str, c: ClaimWitness
|
|
151
|
+
) -> None:
|
|
152
|
+
cur.execute(
|
|
153
|
+
"""
|
|
154
|
+
INSERT OR REPLACE INTO claims (
|
|
155
|
+
claim_id, scope_id, dossier, kind, tag, statement,
|
|
156
|
+
source_hash, source_url, verbatim_quote, parent_claims,
|
|
157
|
+
deductive_logic, falsification, query, finding,
|
|
158
|
+
severity, tier, confidence, status
|
|
159
|
+
) VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)
|
|
160
|
+
""",
|
|
161
|
+
(
|
|
162
|
+
c.claim_id,
|
|
163
|
+
scope_id,
|
|
164
|
+
dossier,
|
|
165
|
+
c.kind,
|
|
166
|
+
c.tag,
|
|
167
|
+
c.statement,
|
|
168
|
+
c.source_hash,
|
|
169
|
+
c.source_url,
|
|
170
|
+
c.verbatim_quote,
|
|
171
|
+
json.dumps(c.parent_claims),
|
|
172
|
+
c.deductive_logic,
|
|
173
|
+
c.falsification,
|
|
174
|
+
c.query,
|
|
175
|
+
c.finding,
|
|
176
|
+
c.severity,
|
|
177
|
+
c.tier,
|
|
178
|
+
c.confidence,
|
|
179
|
+
c.status,
|
|
180
|
+
),
|
|
181
|
+
)
|
|
182
|
+
if c.source_hash:
|
|
183
|
+
cur.execute(
|
|
184
|
+
"""
|
|
185
|
+
INSERT OR IGNORE INTO claim_sources (scope_id, dossier, claim_id, source_hash)
|
|
186
|
+
VALUES (?,?,?,?)
|
|
187
|
+
""",
|
|
188
|
+
(scope_id, dossier, c.claim_id, c.source_hash),
|
|
189
|
+
)
|
|
190
|
+
|
|
191
|
+
def _sync_fts(self) -> None:
|
|
192
|
+
"""Rebuild the FTS index from `claims` (content= external table).
|
|
193
|
+
|
|
194
|
+
Must run after any write batch: the external-content FTS5 table does
|
|
195
|
+
not auto-populate on INSERT into the content table.
|
|
196
|
+
"""
|
|
197
|
+
if not self.fts5:
|
|
198
|
+
return
|
|
199
|
+
self.conn.execute("INSERT INTO claims_fts(claims_fts) VALUES('rebuild')")
|
|
200
|
+
self.conn.commit()
|
|
201
|
+
|
|
202
|
+
def index_source_meta(self, source_hash: str, meta: Dict[str, Any]) -> None:
|
|
203
|
+
self.conn.execute(
|
|
204
|
+
"""
|
|
205
|
+
INSERT OR REPLACE INTO sources (source_hash, url, title, tier, byte_size, char_count)
|
|
206
|
+
VALUES (?,?,?,?,?,?)
|
|
207
|
+
""",
|
|
208
|
+
(
|
|
209
|
+
source_hash,
|
|
210
|
+
meta.get("url"),
|
|
211
|
+
meta.get("title"),
|
|
212
|
+
meta.get("tier"),
|
|
213
|
+
meta.get("byte_size"),
|
|
214
|
+
meta.get("char_count"),
|
|
215
|
+
),
|
|
216
|
+
)
|
|
217
|
+
self.conn.commit()
|
|
218
|
+
|
|
219
|
+
def record_status_event(
|
|
220
|
+
self,
|
|
221
|
+
scope_id: str,
|
|
222
|
+
claim_id: str,
|
|
223
|
+
from_status: str,
|
|
224
|
+
to_status: str,
|
|
225
|
+
reason: str,
|
|
226
|
+
at: str,
|
|
227
|
+
) -> None:
|
|
228
|
+
self.conn.execute(
|
|
229
|
+
"""
|
|
230
|
+
INSERT INTO status_events (scope_id, claim_id, from_status, to_status, reason, at)
|
|
231
|
+
VALUES (?,?,?,?,?,?)
|
|
232
|
+
""",
|
|
233
|
+
(scope_id, claim_id, from_status, to_status, reason, at),
|
|
234
|
+
)
|
|
235
|
+
self.conn.commit()
|
|
236
|
+
|
|
237
|
+
# ---- queries --------------------------------------------------------
|
|
238
|
+
|
|
239
|
+
def search_statements(self, text: str, limit: int = 50) -> List[Dict[str, Any]]:
|
|
240
|
+
if self.fts5:
|
|
241
|
+
rows = self.conn.execute(
|
|
242
|
+
"""
|
|
243
|
+
SELECT c.* FROM claims c
|
|
244
|
+
JOIN claims_fts f ON f.rowid = c.rowid
|
|
245
|
+
WHERE claims_fts MATCH ? LIMIT ?
|
|
246
|
+
""",
|
|
247
|
+
(text, limit),
|
|
248
|
+
).fetchall()
|
|
249
|
+
if rows:
|
|
250
|
+
return [dict(r) for r in rows]
|
|
251
|
+
rows = self.conn.execute(
|
|
252
|
+
"SELECT * FROM claims WHERE statement LIKE ? LIMIT ?",
|
|
253
|
+
(f"%{text}%", limit),
|
|
254
|
+
).fetchall()
|
|
255
|
+
return [dict(r) for r in rows]
|
|
256
|
+
|
|
257
|
+
def claims_citing(self, source_hash: str) -> List[Dict[str, Any]]:
|
|
258
|
+
rows = self.conn.execute(
|
|
259
|
+
"SELECT * FROM claims WHERE source_hash = ?", (source_hash,)
|
|
260
|
+
).fetchall()
|
|
261
|
+
return [dict(r) for r in rows]
|
|
262
|
+
|
|
263
|
+
def content_fingerprint(self) -> str:
|
|
264
|
+
"""Deterministic hash of the claims+sources payload (idempotency probe)."""
|
|
265
|
+
rows = self.conn.execute(
|
|
266
|
+
"""
|
|
267
|
+
SELECT claim_id, scope_id, dossier, kind, tag, statement, source_hash, status
|
|
268
|
+
FROM claims ORDER BY scope_id, dossier, claim_id
|
|
269
|
+
"""
|
|
270
|
+
).fetchall()
|
|
271
|
+
payload = "|".join(
|
|
272
|
+
f"{r['scope_id']}:{r['dossier']}:{r['claim_id']}:{r['tag']}:{r['status']}"
|
|
273
|
+
for r in rows
|
|
274
|
+
)
|
|
275
|
+
return hashlib.sha256(payload.encode("utf-8")).hexdigest()
|
|
276
|
+
|
|
277
|
+
def close(self) -> None:
|
|
278
|
+
self.conn.close()
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
# ---- reindex (lossless from flat files) ---------------------------------
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def reindex(base_dir: Path, store: Optional[ClaimStore] = None) -> Dict[str, Any]:
|
|
285
|
+
"""Rebuild the index entirely from .research/**/alpha|beta_dossier.json
|
|
286
|
+
plus .research/sources/*.json. Idempotent and lossless by contract.
|
|
287
|
+
"""
|
|
288
|
+
base_dir = Path(base_dir)
|
|
289
|
+
own_store = store is None
|
|
290
|
+
if store is None:
|
|
291
|
+
store = ClaimStore(base_dir)
|
|
292
|
+
|
|
293
|
+
dossier_files = sorted(base_dir.glob("scratchpads/*/alpha_dossier.json")) + sorted(
|
|
294
|
+
base_dir.glob("scratchpads/*/beta_dossier.json")
|
|
295
|
+
)
|
|
296
|
+
# Also accept flat scratchpad layouts used by tests
|
|
297
|
+
if not dossier_files:
|
|
298
|
+
dossier_files = sorted(base_dir.glob("**/alpha_dossier.json")) + sorted(
|
|
299
|
+
base_dir.glob("**/beta_dossier.json")
|
|
300
|
+
)
|
|
301
|
+
|
|
302
|
+
n_claims = 0
|
|
303
|
+
for path in dossier_files:
|
|
304
|
+
scope_id = path.parent.name
|
|
305
|
+
try:
|
|
306
|
+
with open(path, "r", encoding="utf-8") as f:
|
|
307
|
+
dossier = json.load(f)
|
|
308
|
+
except (OSError, json.JSONDecodeError):
|
|
309
|
+
continue
|
|
310
|
+
n_claims += store.index_dossier(scope_id, path.name, dossier)
|
|
311
|
+
|
|
312
|
+
n_sources = 0
|
|
313
|
+
for meta_path in sorted((base_dir / "sources").glob("*.json")):
|
|
314
|
+
try:
|
|
315
|
+
with open(meta_path, "r", encoding="utf-8") as f:
|
|
316
|
+
meta = json.load(f)
|
|
317
|
+
except (OSError, json.JSONDecodeError):
|
|
318
|
+
continue
|
|
319
|
+
source_hash = meta.get("hash") or meta_path.stem
|
|
320
|
+
store.index_source_meta(source_hash, meta)
|
|
321
|
+
n_sources += 1
|
|
322
|
+
|
|
323
|
+
if own_store:
|
|
324
|
+
fingerprint = store.content_fingerprint()
|
|
325
|
+
store.close()
|
|
326
|
+
else:
|
|
327
|
+
fingerprint = store.content_fingerprint()
|
|
328
|
+
|
|
329
|
+
return {
|
|
330
|
+
"dossiers_indexed": len(dossier_files),
|
|
331
|
+
"claims_indexed": n_claims,
|
|
332
|
+
"sources_indexed": n_sources,
|
|
333
|
+
"fingerprint": fingerprint,
|
|
334
|
+
"db_path": str(store.db_path),
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
def main(argv: Optional[List[str]] = None) -> int:
|
|
339
|
+
parser = argparse.ArgumentParser(description="IUMBTEMS derived claim index")
|
|
340
|
+
parser.add_argument("--dir", default=".research", help="Path to .research workspace")
|
|
341
|
+
sub = parser.add_subparsers(dest="command")
|
|
342
|
+
sub.add_parser("reindex", help="Rebuild claims.sqlite from flat files")
|
|
343
|
+
search_p = sub.add_parser("search", help="Search claim statements")
|
|
344
|
+
search_p.add_argument("text")
|
|
345
|
+
args = parser.parse_args(argv)
|
|
346
|
+
|
|
347
|
+
base = Path(args.dir)
|
|
348
|
+
if args.command == "reindex" or args.command is None:
|
|
349
|
+
print(json.dumps(reindex(base), indent=2))
|
|
350
|
+
return 0
|
|
351
|
+
if args.command == "search":
|
|
352
|
+
store = ClaimStore(base)
|
|
353
|
+
rows = store.search_statements(args.text)
|
|
354
|
+
store.close()
|
|
355
|
+
print(json.dumps(rows, indent=2, default=str))
|
|
356
|
+
return 0
|
|
357
|
+
return 0
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
if __name__ == "__main__":
|
|
361
|
+
sys.exit(main())
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
ClaimWitness: the single normalized claim record for IUMBTEMS.
|
|
4
|
+
|
|
5
|
+
Dossiers already converge on four shapes (see runner/research_swarm.py mock
|
|
6
|
+
bodies and prompts/*.md):
|
|
7
|
+
affirmative_claims / falsification_claims
|
|
8
|
+
{claim_id, tag, statement, source_hash, source_url, verbatim_quote, [severity]}
|
|
9
|
+
inferred_implications
|
|
10
|
+
{inference_id, tag, statement, parent_claims, deductive_logic, [falsification]}
|
|
11
|
+
hypotheses (beta)
|
|
12
|
+
{hypothesis_id?, claim_id?, tag, statement, falsification, ...}
|
|
13
|
+
negative_knowledge
|
|
14
|
+
{query, finding}
|
|
15
|
+
|
|
16
|
+
This module folds all four into one record so Living Dossiers (staleness),
|
|
17
|
+
PCRB export, Domain Packs, and the refinement checker share one verification
|
|
18
|
+
stack instead of growing four parallel ones.
|
|
19
|
+
|
|
20
|
+
`SourceHasher.verify_quote` remains the ONLY quote engine — `witness_check`
|
|
21
|
+
is a thin adapter over it.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from dataclasses import dataclass, field, asdict
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
from typing import Any, Dict, List, Optional, Tuple
|
|
27
|
+
|
|
28
|
+
# Claim kinds
|
|
29
|
+
KIND_CLAIM = "CLAIM"
|
|
30
|
+
KIND_INFERENCE = "INFERENCE"
|
|
31
|
+
KIND_HYPOTHESIS = "HYPOTHESIS"
|
|
32
|
+
KIND_NEGATIVE_KNOWLEDGE = "NEGATIVE_KNOWLEDGE"
|
|
33
|
+
|
|
34
|
+
# Epistemic tags
|
|
35
|
+
TAG_VERIFIED = "VERIFIED"
|
|
36
|
+
TAG_INFERRED = "INFERRED"
|
|
37
|
+
TAG_HYPOTHESIS = "HYPOTHESIS"
|
|
38
|
+
TAG_NEGATIVE_KNOWLEDGE = "NEGATIVE_KNOWLEDGE"
|
|
39
|
+
|
|
40
|
+
# Living-dossier status (Stream C fills STALE; REJECTED set by the auditor)
|
|
41
|
+
STATUS_LIVE = "LIVE"
|
|
42
|
+
STATUS_STALE = "STALE"
|
|
43
|
+
STATUS_SUSPECT = "SUSPECT"
|
|
44
|
+
STATUS_REJECTED = "REJECTED"
|
|
45
|
+
|
|
46
|
+
# Dossier keys -> (kind, id-field)
|
|
47
|
+
SOURCE_KEYS = (
|
|
48
|
+
("affirmative_claims", KIND_CLAIM, "claim_id"),
|
|
49
|
+
("falsification_claims", KIND_CLAIM, "claim_id"),
|
|
50
|
+
("inferred_implications", KIND_INFERENCE, "inference_id"),
|
|
51
|
+
("hypotheses", KIND_HYPOTHESIS, "hypothesis_id"),
|
|
52
|
+
("negative_knowledge", KIND_NEGATIVE_KNOWLEDGE, None),
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@dataclass
|
|
57
|
+
class ClaimWitness:
|
|
58
|
+
"""One epistemically-tagged assertion in normalized form."""
|
|
59
|
+
|
|
60
|
+
claim_id: str
|
|
61
|
+
kind: str
|
|
62
|
+
tag: str
|
|
63
|
+
statement: str
|
|
64
|
+
|
|
65
|
+
# CLAIM fields (verbatim evidence)
|
|
66
|
+
source_hash: Optional[str] = None
|
|
67
|
+
source_url: Optional[str] = None
|
|
68
|
+
verbatim_quote: Optional[str] = None
|
|
69
|
+
|
|
70
|
+
# INFERENCE / HYPOTHESIS fields
|
|
71
|
+
parent_claims: List[str] = field(default_factory=list)
|
|
72
|
+
deductive_logic: Optional[str] = None
|
|
73
|
+
falsification: Optional[str] = None
|
|
74
|
+
|
|
75
|
+
# NEGATIVE_KNOWLEDGE fields
|
|
76
|
+
query: Optional[str] = None
|
|
77
|
+
finding: Optional[str] = None
|
|
78
|
+
|
|
79
|
+
# Beta adversarial claims
|
|
80
|
+
severity: Optional[str] = None
|
|
81
|
+
|
|
82
|
+
# Filled by the auditor / claim store
|
|
83
|
+
tier: Optional[str] = None
|
|
84
|
+
confidence: Optional[float] = None
|
|
85
|
+
verified_at: Optional[str] = None
|
|
86
|
+
|
|
87
|
+
# Living-dossier slot (Stream C)
|
|
88
|
+
status: str = STATUS_LIVE
|
|
89
|
+
|
|
90
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
91
|
+
return asdict(self)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def normalize_claim(raw: Dict[str, Any], kind: str = KIND_CLAIM) -> ClaimWitness:
|
|
95
|
+
"""Fold one raw dossier entry into a ClaimWitness.
|
|
96
|
+
|
|
97
|
+
`tag` is read from the entry and defaulted by kind; kind wins when the
|
|
98
|
+
entry's tag contradicts its source key (e.g. a HYPOTHESIS-tagged row
|
|
99
|
+
inside inferred_implications is still an INFERENCE for invariant checks).
|
|
100
|
+
"""
|
|
101
|
+
raw_tag = raw.get("tag")
|
|
102
|
+
# A missing tag (key absent) defaults by kind; an EXPLICIT empty tag stays
|
|
103
|
+
# empty so Domain Packs can flag unclassified claims (MISSING_TAG).
|
|
104
|
+
if raw_tag is None:
|
|
105
|
+
tag = {
|
|
106
|
+
KIND_CLAIM: TAG_VERIFIED,
|
|
107
|
+
KIND_INFERENCE: TAG_INFERRED,
|
|
108
|
+
KIND_HYPOTHESIS: TAG_HYPOTHESIS,
|
|
109
|
+
KIND_NEGATIVE_KNOWLEDGE: TAG_NEGATIVE_KNOWLEDGE,
|
|
110
|
+
}[kind]
|
|
111
|
+
else:
|
|
112
|
+
tag = str(raw_tag).upper()
|
|
113
|
+
id_field = {
|
|
114
|
+
KIND_CLAIM: "claim_id",
|
|
115
|
+
KIND_INFERENCE: "inference_id",
|
|
116
|
+
KIND_HYPOTHESIS: "hypothesis_id",
|
|
117
|
+
}.get(kind)
|
|
118
|
+
|
|
119
|
+
claim_id = (
|
|
120
|
+
raw.get("claim_id")
|
|
121
|
+
or raw.get("inference_id")
|
|
122
|
+
or raw.get("hypothesis_id")
|
|
123
|
+
or (id_field and raw.get(id_field))
|
|
124
|
+
or "UNKNOWN"
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
parent_claims = raw.get("parent_claims") or []
|
|
128
|
+
if isinstance(parent_claims, str):
|
|
129
|
+
parent_claims = [parent_claims]
|
|
130
|
+
|
|
131
|
+
return ClaimWitness(
|
|
132
|
+
claim_id=str(claim_id),
|
|
133
|
+
kind=kind,
|
|
134
|
+
tag=tag,
|
|
135
|
+
statement=raw.get("statement") or raw.get("finding") or raw.get("query") or "",
|
|
136
|
+
source_hash=raw.get("source_hash"),
|
|
137
|
+
source_url=raw.get("source_url"),
|
|
138
|
+
verbatim_quote=raw.get("verbatim_quote"),
|
|
139
|
+
parent_claims=[str(p) for p in parent_claims],
|
|
140
|
+
deductive_logic=raw.get("deductive_logic"),
|
|
141
|
+
falsification=raw.get("falsification"),
|
|
142
|
+
query=raw.get("query"),
|
|
143
|
+
finding=raw.get("finding"),
|
|
144
|
+
severity=raw.get("severity"),
|
|
145
|
+
tier=raw.get("tier"),
|
|
146
|
+
confidence=raw.get("confidence"),
|
|
147
|
+
verified_at=raw.get("verified_at"),
|
|
148
|
+
status=raw.get("status") or STATUS_LIVE,
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def claims_from_dossier(dossier: Dict[str, Any]) -> List[ClaimWitness]:
|
|
153
|
+
"""Extract every ClaimWitness from one alpha/beta dossier dict."""
|
|
154
|
+
out: List[ClaimWitness] = []
|
|
155
|
+
for key, kind, _id_field in SOURCE_KEYS:
|
|
156
|
+
rows = dossier.get(key) or []
|
|
157
|
+
if not isinstance(rows, list):
|
|
158
|
+
continue
|
|
159
|
+
for row in rows:
|
|
160
|
+
if isinstance(row, dict):
|
|
161
|
+
out.append(normalize_claim(row, kind=kind))
|
|
162
|
+
return out
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def witness_check(
|
|
166
|
+
claim: ClaimWitness, hasher: Any
|
|
167
|
+
) -> Tuple[bool, float, Optional[str]]:
|
|
168
|
+
"""Single verification stack: delegate to SourceHasher.verify_quote.
|
|
169
|
+
|
|
170
|
+
Returns (is_verified, confidence, message). Claims with no quote or no
|
|
171
|
+
hash are never verified (the auditor rejects them as UNVERIFIED_REJECTED).
|
|
172
|
+
"""
|
|
173
|
+
if not claim.source_hash or not claim.verbatim_quote:
|
|
174
|
+
return False, 0.0, "Missing source_hash or verbatim_quote"
|
|
175
|
+
return hasher.verify_quote(claim.source_hash, claim.verbatim_quote)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def load_dossier(path: Path) -> Dict[str, Any]:
|
|
179
|
+
"""Load an alpha_dossier.json / beta_dossier.json file."""
|
|
180
|
+
import json
|
|
181
|
+
|
|
182
|
+
with open(path, "r", encoding="utf-8") as f:
|
|
183
|
+
return json.load(f)
|