@heretek-ai/epistemic-swarm 0.2.2 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (138) hide show
  1. package/.agents/skills/brainstorming/SKILL.md +13 -0
  2. package/.agents/skills/code_audit/SKILL.md +13 -0
  3. package/.agents/skills/epistemic_search/SKILL.md +13 -0
  4. package/.agents/skills/grilling/SKILL.md +13 -0
  5. package/.agents/skills/oss_scout/SKILL.md +13 -0
  6. package/.agents/skills/research_cache/SKILL.md +13 -0
  7. package/.agents/skills/swarm_config/SKILL.md +13 -0
  8. package/.claude-plugin/plugin.json +15 -5
  9. package/.omp/README.md +39 -0
  10. package/.omp/SYSTEM.md +12 -0
  11. package/.omp/commands/audit.md +10 -0
  12. package/.omp/commands/brainstorming.md +12 -0
  13. package/.omp/commands/grill.md +9 -0
  14. package/.omp/commands/scout.md +10 -0
  15. package/.omp/commands/swarm-config.md +10 -0
  16. package/.omp/commands/swarm.md +9 -0
  17. package/.omp/hooks/post/epistemic-audit.ts +16 -0
  18. package/.omp/hooks/pre/epistemic-redirect.ts +21 -0
  19. package/.omp/prompts/brainstorming.md +8 -0
  20. package/.omp/prompts/swarm.md +8 -0
  21. package/MARKETPLACE.md +8 -0
  22. package/README.md +53 -8
  23. package/bin/cli.js +27 -3
  24. package/config/domain_packs/biopharma.json +23 -0
  25. package/config/domain_packs/legal.json +19 -0
  26. package/config/domain_packs/quant.json +19 -0
  27. package/config/mcp-research-servers.json +7 -0
  28. package/config/mcp_launcher.py +48 -137
  29. package/config/opencode-snippet.json +58 -3
  30. package/config/searxng_mcp.py +42 -83
  31. package/extensions/pi/index.js +196 -28
  32. package/install.sh +20 -4
  33. package/package.json +40 -5
  34. package/plugins/antigravity/README.md +28 -0
  35. package/plugins/antigravity/agents/alpha-thesis.md +6 -0
  36. package/plugins/antigravity/agents/beta-antithesis.md +7 -0
  37. package/plugins/antigravity/agents/brainstormer.md +7 -0
  38. package/plugins/antigravity/agents/epistemic-auditor.md +5 -0
  39. package/plugins/antigravity/hooks.json +23 -0
  40. package/plugins/antigravity/mcp_config.json +33 -0
  41. package/plugins/antigravity/plugin.json +21 -0
  42. package/plugins/antigravity/rules/epistemic-integrity.md +6 -0
  43. package/plugins/antigravity/skills/brainstorming/SKILL.md +13 -0
  44. package/plugins/antigravity/skills/code_audit/SKILL.md +13 -0
  45. package/plugins/antigravity/skills/epistemic_search/SKILL.md +13 -0
  46. package/plugins/antigravity/skills/grilling/SKILL.md +13 -0
  47. package/plugins/antigravity/skills/oss_scout/SKILL.md +13 -0
  48. package/plugins/antigravity/skills/research_cache/SKILL.md +13 -0
  49. package/plugins/antigravity/skills/swarm_config/SKILL.md +13 -0
  50. package/plugins/codex/AGENTS.md.snippet +10 -0
  51. package/plugins/codex/README.md +37 -0
  52. package/plugins/codex/config.toml.snippet +28 -0
  53. package/plugins/codex/openai.yaml +24 -0
  54. package/plugins/codex/skills/brainstorming/SKILL.md +13 -0
  55. package/plugins/codex/skills/code_audit/SKILL.md +13 -0
  56. package/plugins/codex/skills/epistemic_search/SKILL.md +13 -0
  57. package/plugins/codex/skills/grilling/SKILL.md +13 -0
  58. package/plugins/codex/skills/oss_scout/SKILL.md +13 -0
  59. package/plugins/codex/skills/research_cache/SKILL.md +13 -0
  60. package/plugins/codex/skills/swarm_config/SKILL.md +13 -0
  61. package/plugins/gemini/GEMINI.md +15 -0
  62. package/plugins/gemini/README.md +19 -0
  63. package/plugins/gemini/commands/audit.toml +6 -0
  64. package/plugins/gemini/commands/brainstorming.toml +10 -0
  65. package/plugins/gemini/commands/grill.toml +6 -0
  66. package/plugins/gemini/commands/scout.toml +7 -0
  67. package/plugins/gemini/commands/swarm-config.toml +7 -0
  68. package/plugins/gemini/commands/swarm.toml +8 -0
  69. package/plugins/gemini/gemini-extension.json +38 -0
  70. package/plugins/gemini/hooks/hooks.json +11 -0
  71. package/plugins/gemini/skills/brainstorming/SKILL.md +13 -0
  72. package/plugins/gemini/skills/code_audit/SKILL.md +13 -0
  73. package/plugins/gemini/skills/epistemic_search/SKILL.md +13 -0
  74. package/plugins/gemini/skills/grilling/SKILL.md +13 -0
  75. package/plugins/gemini/skills/oss_scout/SKILL.md +13 -0
  76. package/plugins/gemini/skills/research_cache/SKILL.md +13 -0
  77. package/plugins/gemini/skills/swarm_config/SKILL.md +13 -0
  78. package/plugins/opencode/index.js +335 -118
  79. package/prompts/agent_brainstormer.md +97 -0
  80. package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
  81. package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
  82. package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
  83. package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
  84. package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
  85. package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
  86. package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
  87. package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
  88. package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
  89. package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
  90. package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
  91. package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
  92. package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
  93. package/runner/auctioneer.py +169 -0
  94. package/runner/auditor_engine.py +127 -12
  95. package/runner/claim_store.py +361 -0
  96. package/runner/claim_witness.py +183 -0
  97. package/runner/living_dossiers.py +355 -0
  98. package/runner/mcp_protocol.py +188 -0
  99. package/runner/mcp_server.py +567 -0
  100. package/runner/pcrb.py +212 -0
  101. package/runner/pcrb_verify.py +237 -0
  102. package/runner/refinement.py +335 -0
  103. package/runner/research_swarm.py +378 -94
  104. package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
  105. package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
  106. package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
  107. package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
  108. package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
  109. package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
  110. package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
  111. package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
  112. package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
  113. package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
  114. package/runner/tests/fixtures/auction_objective.json +35 -0
  115. package/runner/tests/fixtures/borderline_claims.json +27 -0
  116. package/runner/tests/fixtures/divergence_objectives.json +31 -0
  117. package/runner/tests/test_auction_order.py +173 -0
  118. package/runner/tests/test_claim_store.py +162 -0
  119. package/runner/tests/test_claim_witness.py +186 -0
  120. package/runner/tests/test_domain_packs.py +217 -0
  121. package/runner/tests/test_fleet_seam.py +113 -0
  122. package/runner/tests/test_living_dossiers.py +204 -0
  123. package/runner/tests/test_mcp_server.py +212 -0
  124. package/runner/tests/test_pcrb.py +240 -0
  125. package/runner/tests/test_refinement.py +255 -0
  126. package/runner/tests/test_swarm.py +173 -15
  127. package/scripts/auction_experiment.py +180 -0
  128. package/scripts/build_adapters.py +183 -0
  129. package/scripts/divergence_experiment.py +184 -0
  130. package/skills/brainstorming/SKILL.md +106 -0
  131. package/skills/brainstorming/__init__.py +1 -0
  132. package/skills/brainstorming/scripts/brainstorm.py +200 -0
  133. package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
  134. package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
  135. package/skills/swarm_config/SKILL.md +1 -1
  136. package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
  137. package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
  138. package/skills/swarm_config/configure.py +45 -11
@@ -0,0 +1,361 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ Derived claim index over the .research/ flat-file record.
4
+
5
+ INVIOLABLE: .research/**/*.json|md stays the source of truth. This sqlite
6
+ DB is a rebuildable index for cross-session dedup, degradation queries, and
7
+ provenance joins. If it ever drifts: delete it and `--reindex`. Never
8
+ "fix the DB" in place.
9
+
10
+ Tables:
11
+ claims one row per ClaimWitness
12
+ sources one row per content hash (from .research/sources/<hash>.json)
13
+ claim_sources join (claim_id -> source_hash)
14
+ status_events append-only degradation log (Stream C mirrors into here)
15
+
16
+ Zero external deps: stdlib sqlite3. FTS5 over `statement` when compiled in,
17
+ else a LIKE scan. An `embedding BLOB` column is reserved for the vector half
18
+ of the dual-layer design; embeddings are deliberately NOT built here (they
19
+ need an API call and are out of this phase's scope).
20
+ """
21
+
22
+ import argparse
23
+ import hashlib
24
+ import json
25
+ import sqlite3
26
+ import sys
27
+ from pathlib import Path
28
+ from typing import Any, Dict, Iterable, List, Optional, Tuple
29
+
30
+ SCRIPT_DIR = Path(__file__).resolve().parent
31
+ PROJECT_ROOT = SCRIPT_DIR.parent
32
+ if str(PROJECT_ROOT) not in sys.path:
33
+ sys.path.insert(0, str(PROJECT_ROOT))
34
+
35
+ from runner.claim_witness import ClaimWitness, claims_from_dossier # noqa: E402
36
+
37
+ SCHEMA_VERSION = 1
38
+
39
+ DOSIER_NAMES = ("alpha_dossier.json", "beta_dossier.json")
40
+
41
+
42
+ def _has_fts5(conn: sqlite3.Connection) -> bool:
43
+ try:
44
+ conn.execute("CREATE VIRTUAL TABLE IF NOT EXISTS _fts5_probe USING fts5(x)")
45
+ conn.execute("DROP TABLE _fts5_probe")
46
+ return True
47
+ except sqlite3.OperationalError:
48
+ return False
49
+
50
+
51
+ class ClaimStore:
52
+ def __init__(self, base_dir: Path, db_path: Optional[Path] = None):
53
+ self.base_dir = Path(base_dir)
54
+ self.db_path = db_path or (self.base_dir / "claims.sqlite")
55
+ self.db_path.parent.mkdir(parents=True, exist_ok=True)
56
+ self.conn = sqlite3.connect(str(self.db_path))
57
+ self.conn.row_factory = sqlite3.Row
58
+ self.fts5 = _has_fts5(self.conn)
59
+ self._create_schema()
60
+
61
+ def _create_schema(self) -> None:
62
+ cur = self.conn.cursor()
63
+ cur.executescript(
64
+ """
65
+ CREATE TABLE IF NOT EXISTS meta (
66
+ key TEXT PRIMARY KEY,
67
+ value TEXT NOT NULL
68
+ );
69
+ CREATE TABLE IF NOT EXISTS sources (
70
+ source_hash TEXT PRIMARY KEY,
71
+ url TEXT,
72
+ title TEXT,
73
+ tier TEXT,
74
+ byte_size INTEGER,
75
+ char_count INTEGER
76
+ );
77
+ CREATE TABLE IF NOT EXISTS claims (
78
+ claim_id TEXT NOT NULL,
79
+ scope_id TEXT,
80
+ dossier TEXT,
81
+ kind TEXT,
82
+ tag TEXT,
83
+ statement TEXT,
84
+ source_hash TEXT,
85
+ source_url TEXT,
86
+ verbatim_quote TEXT,
87
+ parent_claims TEXT,
88
+ deductive_logic TEXT,
89
+ falsification TEXT,
90
+ query TEXT,
91
+ finding TEXT,
92
+ severity TEXT,
93
+ tier TEXT,
94
+ confidence REAL,
95
+ status TEXT,
96
+ embedding BLOB,
97
+ PRIMARY KEY (scope_id, dossier, claim_id)
98
+ );
99
+ CREATE TABLE IF NOT EXISTS claim_sources (
100
+ scope_id TEXT,
101
+ dossier TEXT,
102
+ claim_id TEXT,
103
+ source_hash TEXT,
104
+ PRIMARY KEY (scope_id, dossier, claim_id, source_hash)
105
+ );
106
+ CREATE TABLE IF NOT EXISTS status_events (
107
+ event_id INTEGER PRIMARY KEY AUTOINCREMENT,
108
+ scope_id TEXT,
109
+ claim_id TEXT,
110
+ from_status TEXT,
111
+ to_status TEXT,
112
+ reason TEXT,
113
+ at TEXT
114
+ );
115
+ """
116
+ )
117
+ cur.execute("INSERT OR IGNORE INTO meta(key, value) VALUES ('schema_version', ?)", (str(SCHEMA_VERSION),))
118
+ if self.fts5:
119
+ cur.execute(
120
+ """
121
+ CREATE VIRTUAL TABLE IF NOT EXISTS claims_fts
122
+ USING fts5(claim_id, statement, content='claims', content_rowid='rowid')
123
+ """
124
+ )
125
+ self.conn.commit()
126
+
127
+ # ---- indexing -------------------------------------------------------
128
+
129
+ def index_dossier(
130
+ self, scope_id: str, dossier_name: str, dossier: Dict[str, Any]
131
+ ) -> int:
132
+ """Index one dossier dict. Idempotent: re-indexing replaces rows."""
133
+ claims = claims_from_dossier(dossier)
134
+ cur = self.conn.cursor()
135
+ cur.execute(
136
+ "DELETE FROM claims WHERE scope_id = ? AND dossier = ?",
137
+ (scope_id, dossier_name),
138
+ )
139
+ cur.execute(
140
+ "DELETE FROM claim_sources WHERE scope_id = ? AND dossier = ?",
141
+ (scope_id, dossier_name),
142
+ )
143
+ for c in claims:
144
+ self._upsert_claim(cur, scope_id, dossier_name, c)
145
+ self.conn.commit()
146
+ self._sync_fts()
147
+ return len(claims)
148
+
149
+ def _upsert_claim(
150
+ self, cur: sqlite3.Cursor, scope_id: str, dossier: str, c: ClaimWitness
151
+ ) -> None:
152
+ cur.execute(
153
+ """
154
+ INSERT OR REPLACE INTO claims (
155
+ claim_id, scope_id, dossier, kind, tag, statement,
156
+ source_hash, source_url, verbatim_quote, parent_claims,
157
+ deductive_logic, falsification, query, finding,
158
+ severity, tier, confidence, status
159
+ ) VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)
160
+ """,
161
+ (
162
+ c.claim_id,
163
+ scope_id,
164
+ dossier,
165
+ c.kind,
166
+ c.tag,
167
+ c.statement,
168
+ c.source_hash,
169
+ c.source_url,
170
+ c.verbatim_quote,
171
+ json.dumps(c.parent_claims),
172
+ c.deductive_logic,
173
+ c.falsification,
174
+ c.query,
175
+ c.finding,
176
+ c.severity,
177
+ c.tier,
178
+ c.confidence,
179
+ c.status,
180
+ ),
181
+ )
182
+ if c.source_hash:
183
+ cur.execute(
184
+ """
185
+ INSERT OR IGNORE INTO claim_sources (scope_id, dossier, claim_id, source_hash)
186
+ VALUES (?,?,?,?)
187
+ """,
188
+ (scope_id, dossier, c.claim_id, c.source_hash),
189
+ )
190
+
191
+ def _sync_fts(self) -> None:
192
+ """Rebuild the FTS index from `claims` (content= external table).
193
+
194
+ Must run after any write batch: the external-content FTS5 table does
195
+ not auto-populate on INSERT into the content table.
196
+ """
197
+ if not self.fts5:
198
+ return
199
+ self.conn.execute("INSERT INTO claims_fts(claims_fts) VALUES('rebuild')")
200
+ self.conn.commit()
201
+
202
+ def index_source_meta(self, source_hash: str, meta: Dict[str, Any]) -> None:
203
+ self.conn.execute(
204
+ """
205
+ INSERT OR REPLACE INTO sources (source_hash, url, title, tier, byte_size, char_count)
206
+ VALUES (?,?,?,?,?,?)
207
+ """,
208
+ (
209
+ source_hash,
210
+ meta.get("url"),
211
+ meta.get("title"),
212
+ meta.get("tier"),
213
+ meta.get("byte_size"),
214
+ meta.get("char_count"),
215
+ ),
216
+ )
217
+ self.conn.commit()
218
+
219
+ def record_status_event(
220
+ self,
221
+ scope_id: str,
222
+ claim_id: str,
223
+ from_status: str,
224
+ to_status: str,
225
+ reason: str,
226
+ at: str,
227
+ ) -> None:
228
+ self.conn.execute(
229
+ """
230
+ INSERT INTO status_events (scope_id, claim_id, from_status, to_status, reason, at)
231
+ VALUES (?,?,?,?,?,?)
232
+ """,
233
+ (scope_id, claim_id, from_status, to_status, reason, at),
234
+ )
235
+ self.conn.commit()
236
+
237
+ # ---- queries --------------------------------------------------------
238
+
239
+ def search_statements(self, text: str, limit: int = 50) -> List[Dict[str, Any]]:
240
+ if self.fts5:
241
+ rows = self.conn.execute(
242
+ """
243
+ SELECT c.* FROM claims c
244
+ JOIN claims_fts f ON f.rowid = c.rowid
245
+ WHERE claims_fts MATCH ? LIMIT ?
246
+ """,
247
+ (text, limit),
248
+ ).fetchall()
249
+ if rows:
250
+ return [dict(r) for r in rows]
251
+ rows = self.conn.execute(
252
+ "SELECT * FROM claims WHERE statement LIKE ? LIMIT ?",
253
+ (f"%{text}%", limit),
254
+ ).fetchall()
255
+ return [dict(r) for r in rows]
256
+
257
+ def claims_citing(self, source_hash: str) -> List[Dict[str, Any]]:
258
+ rows = self.conn.execute(
259
+ "SELECT * FROM claims WHERE source_hash = ?", (source_hash,)
260
+ ).fetchall()
261
+ return [dict(r) for r in rows]
262
+
263
+ def content_fingerprint(self) -> str:
264
+ """Deterministic hash of the claims+sources payload (idempotency probe)."""
265
+ rows = self.conn.execute(
266
+ """
267
+ SELECT claim_id, scope_id, dossier, kind, tag, statement, source_hash, status
268
+ FROM claims ORDER BY scope_id, dossier, claim_id
269
+ """
270
+ ).fetchall()
271
+ payload = "|".join(
272
+ f"{r['scope_id']}:{r['dossier']}:{r['claim_id']}:{r['tag']}:{r['status']}"
273
+ for r in rows
274
+ )
275
+ return hashlib.sha256(payload.encode("utf-8")).hexdigest()
276
+
277
+ def close(self) -> None:
278
+ self.conn.close()
279
+
280
+
281
+ # ---- reindex (lossless from flat files) ---------------------------------
282
+
283
+
284
+ def reindex(base_dir: Path, store: Optional[ClaimStore] = None) -> Dict[str, Any]:
285
+ """Rebuild the index entirely from .research/**/alpha|beta_dossier.json
286
+ plus .research/sources/*.json. Idempotent and lossless by contract.
287
+ """
288
+ base_dir = Path(base_dir)
289
+ own_store = store is None
290
+ if store is None:
291
+ store = ClaimStore(base_dir)
292
+
293
+ dossier_files = sorted(base_dir.glob("scratchpads/*/alpha_dossier.json")) + sorted(
294
+ base_dir.glob("scratchpads/*/beta_dossier.json")
295
+ )
296
+ # Also accept flat scratchpad layouts used by tests
297
+ if not dossier_files:
298
+ dossier_files = sorted(base_dir.glob("**/alpha_dossier.json")) + sorted(
299
+ base_dir.glob("**/beta_dossier.json")
300
+ )
301
+
302
+ n_claims = 0
303
+ for path in dossier_files:
304
+ scope_id = path.parent.name
305
+ try:
306
+ with open(path, "r", encoding="utf-8") as f:
307
+ dossier = json.load(f)
308
+ except (OSError, json.JSONDecodeError):
309
+ continue
310
+ n_claims += store.index_dossier(scope_id, path.name, dossier)
311
+
312
+ n_sources = 0
313
+ for meta_path in sorted((base_dir / "sources").glob("*.json")):
314
+ try:
315
+ with open(meta_path, "r", encoding="utf-8") as f:
316
+ meta = json.load(f)
317
+ except (OSError, json.JSONDecodeError):
318
+ continue
319
+ source_hash = meta.get("hash") or meta_path.stem
320
+ store.index_source_meta(source_hash, meta)
321
+ n_sources += 1
322
+
323
+ if own_store:
324
+ fingerprint = store.content_fingerprint()
325
+ store.close()
326
+ else:
327
+ fingerprint = store.content_fingerprint()
328
+
329
+ return {
330
+ "dossiers_indexed": len(dossier_files),
331
+ "claims_indexed": n_claims,
332
+ "sources_indexed": n_sources,
333
+ "fingerprint": fingerprint,
334
+ "db_path": str(store.db_path),
335
+ }
336
+
337
+
338
+ def main(argv: Optional[List[str]] = None) -> int:
339
+ parser = argparse.ArgumentParser(description="IUMBTEMS derived claim index")
340
+ parser.add_argument("--dir", default=".research", help="Path to .research workspace")
341
+ sub = parser.add_subparsers(dest="command")
342
+ sub.add_parser("reindex", help="Rebuild claims.sqlite from flat files")
343
+ search_p = sub.add_parser("search", help="Search claim statements")
344
+ search_p.add_argument("text")
345
+ args = parser.parse_args(argv)
346
+
347
+ base = Path(args.dir)
348
+ if args.command == "reindex" or args.command is None:
349
+ print(json.dumps(reindex(base), indent=2))
350
+ return 0
351
+ if args.command == "search":
352
+ store = ClaimStore(base)
353
+ rows = store.search_statements(args.text)
354
+ store.close()
355
+ print(json.dumps(rows, indent=2, default=str))
356
+ return 0
357
+ return 0
358
+
359
+
360
+ if __name__ == "__main__":
361
+ sys.exit(main())
@@ -0,0 +1,183 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ ClaimWitness: the single normalized claim record for IUMBTEMS.
4
+
5
+ Dossiers already converge on four shapes (see runner/research_swarm.py mock
6
+ bodies and prompts/*.md):
7
+ affirmative_claims / falsification_claims
8
+ {claim_id, tag, statement, source_hash, source_url, verbatim_quote, [severity]}
9
+ inferred_implications
10
+ {inference_id, tag, statement, parent_claims, deductive_logic, [falsification]}
11
+ hypotheses (beta)
12
+ {hypothesis_id?, claim_id?, tag, statement, falsification, ...}
13
+ negative_knowledge
14
+ {query, finding}
15
+
16
+ This module folds all four into one record so Living Dossiers (staleness),
17
+ PCRB export, Domain Packs, and the refinement checker share one verification
18
+ stack instead of growing four parallel ones.
19
+
20
+ `SourceHasher.verify_quote` remains the ONLY quote engine — `witness_check`
21
+ is a thin adapter over it.
22
+ """
23
+
24
+ from dataclasses import dataclass, field, asdict
25
+ from pathlib import Path
26
+ from typing import Any, Dict, List, Optional, Tuple
27
+
28
+ # Claim kinds
29
+ KIND_CLAIM = "CLAIM"
30
+ KIND_INFERENCE = "INFERENCE"
31
+ KIND_HYPOTHESIS = "HYPOTHESIS"
32
+ KIND_NEGATIVE_KNOWLEDGE = "NEGATIVE_KNOWLEDGE"
33
+
34
+ # Epistemic tags
35
+ TAG_VERIFIED = "VERIFIED"
36
+ TAG_INFERRED = "INFERRED"
37
+ TAG_HYPOTHESIS = "HYPOTHESIS"
38
+ TAG_NEGATIVE_KNOWLEDGE = "NEGATIVE_KNOWLEDGE"
39
+
40
+ # Living-dossier status (Stream C fills STALE; REJECTED set by the auditor)
41
+ STATUS_LIVE = "LIVE"
42
+ STATUS_STALE = "STALE"
43
+ STATUS_SUSPECT = "SUSPECT"
44
+ STATUS_REJECTED = "REJECTED"
45
+
46
+ # Dossier keys -> (kind, id-field)
47
+ SOURCE_KEYS = (
48
+ ("affirmative_claims", KIND_CLAIM, "claim_id"),
49
+ ("falsification_claims", KIND_CLAIM, "claim_id"),
50
+ ("inferred_implications", KIND_INFERENCE, "inference_id"),
51
+ ("hypotheses", KIND_HYPOTHESIS, "hypothesis_id"),
52
+ ("negative_knowledge", KIND_NEGATIVE_KNOWLEDGE, None),
53
+ )
54
+
55
+
56
+ @dataclass
57
+ class ClaimWitness:
58
+ """One epistemically-tagged assertion in normalized form."""
59
+
60
+ claim_id: str
61
+ kind: str
62
+ tag: str
63
+ statement: str
64
+
65
+ # CLAIM fields (verbatim evidence)
66
+ source_hash: Optional[str] = None
67
+ source_url: Optional[str] = None
68
+ verbatim_quote: Optional[str] = None
69
+
70
+ # INFERENCE / HYPOTHESIS fields
71
+ parent_claims: List[str] = field(default_factory=list)
72
+ deductive_logic: Optional[str] = None
73
+ falsification: Optional[str] = None
74
+
75
+ # NEGATIVE_KNOWLEDGE fields
76
+ query: Optional[str] = None
77
+ finding: Optional[str] = None
78
+
79
+ # Beta adversarial claims
80
+ severity: Optional[str] = None
81
+
82
+ # Filled by the auditor / claim store
83
+ tier: Optional[str] = None
84
+ confidence: Optional[float] = None
85
+ verified_at: Optional[str] = None
86
+
87
+ # Living-dossier slot (Stream C)
88
+ status: str = STATUS_LIVE
89
+
90
+ def to_dict(self) -> Dict[str, Any]:
91
+ return asdict(self)
92
+
93
+
94
+ def normalize_claim(raw: Dict[str, Any], kind: str = KIND_CLAIM) -> ClaimWitness:
95
+ """Fold one raw dossier entry into a ClaimWitness.
96
+
97
+ `tag` is read from the entry and defaulted by kind; kind wins when the
98
+ entry's tag contradicts its source key (e.g. a HYPOTHESIS-tagged row
99
+ inside inferred_implications is still an INFERENCE for invariant checks).
100
+ """
101
+ raw_tag = raw.get("tag")
102
+ # A missing tag (key absent) defaults by kind; an EXPLICIT empty tag stays
103
+ # empty so Domain Packs can flag unclassified claims (MISSING_TAG).
104
+ if raw_tag is None:
105
+ tag = {
106
+ KIND_CLAIM: TAG_VERIFIED,
107
+ KIND_INFERENCE: TAG_INFERRED,
108
+ KIND_HYPOTHESIS: TAG_HYPOTHESIS,
109
+ KIND_NEGATIVE_KNOWLEDGE: TAG_NEGATIVE_KNOWLEDGE,
110
+ }[kind]
111
+ else:
112
+ tag = str(raw_tag).upper()
113
+ id_field = {
114
+ KIND_CLAIM: "claim_id",
115
+ KIND_INFERENCE: "inference_id",
116
+ KIND_HYPOTHESIS: "hypothesis_id",
117
+ }.get(kind)
118
+
119
+ claim_id = (
120
+ raw.get("claim_id")
121
+ or raw.get("inference_id")
122
+ or raw.get("hypothesis_id")
123
+ or (id_field and raw.get(id_field))
124
+ or "UNKNOWN"
125
+ )
126
+
127
+ parent_claims = raw.get("parent_claims") or []
128
+ if isinstance(parent_claims, str):
129
+ parent_claims = [parent_claims]
130
+
131
+ return ClaimWitness(
132
+ claim_id=str(claim_id),
133
+ kind=kind,
134
+ tag=tag,
135
+ statement=raw.get("statement") or raw.get("finding") or raw.get("query") or "",
136
+ source_hash=raw.get("source_hash"),
137
+ source_url=raw.get("source_url"),
138
+ verbatim_quote=raw.get("verbatim_quote"),
139
+ parent_claims=[str(p) for p in parent_claims],
140
+ deductive_logic=raw.get("deductive_logic"),
141
+ falsification=raw.get("falsification"),
142
+ query=raw.get("query"),
143
+ finding=raw.get("finding"),
144
+ severity=raw.get("severity"),
145
+ tier=raw.get("tier"),
146
+ confidence=raw.get("confidence"),
147
+ verified_at=raw.get("verified_at"),
148
+ status=raw.get("status") or STATUS_LIVE,
149
+ )
150
+
151
+
152
+ def claims_from_dossier(dossier: Dict[str, Any]) -> List[ClaimWitness]:
153
+ """Extract every ClaimWitness from one alpha/beta dossier dict."""
154
+ out: List[ClaimWitness] = []
155
+ for key, kind, _id_field in SOURCE_KEYS:
156
+ rows = dossier.get(key) or []
157
+ if not isinstance(rows, list):
158
+ continue
159
+ for row in rows:
160
+ if isinstance(row, dict):
161
+ out.append(normalize_claim(row, kind=kind))
162
+ return out
163
+
164
+
165
+ def witness_check(
166
+ claim: ClaimWitness, hasher: Any
167
+ ) -> Tuple[bool, float, Optional[str]]:
168
+ """Single verification stack: delegate to SourceHasher.verify_quote.
169
+
170
+ Returns (is_verified, confidence, message). Claims with no quote or no
171
+ hash are never verified (the auditor rejects them as UNVERIFIED_REJECTED).
172
+ """
173
+ if not claim.source_hash or not claim.verbatim_quote:
174
+ return False, 0.0, "Missing source_hash or verbatim_quote"
175
+ return hasher.verify_quote(claim.source_hash, claim.verbatim_quote)
176
+
177
+
178
+ def load_dossier(path: Path) -> Dict[str, Any]:
179
+ """Load an alpha_dossier.json / beta_dossier.json file."""
180
+ import json
181
+
182
+ with open(path, "r", encoding="utf-8") as f:
183
+ return json.load(f)