@heretek-ai/epistemic-swarm 0.7.0 → 0.7.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/bin/cli.js +1 -0
- package/extensions/pi/index.js +7 -1
- package/package.json +1 -1
- package/plugins/darkharvest/skills/darkharvest/scripts/harvest.py +76 -39
- package/plugins/factory/skills/factory/scripts/factory.py +1 -1
- package/plugins/opencode/index.js +41 -1
- package/plugins/opencode/tui.js +66 -38
- package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
- package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
- package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
- package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
- package/runner/__pycache__/path_safety.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
- package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
- package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
- package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
- package/runner/claim_store.py +15 -4
- package/runner/living_dossiers.py +74 -56
- package/runner/mcp_server.py +37 -2
- package/runner/pcrb.py +53 -29
- package/runner/refinement.py +70 -27
- package/runner/research_swarm.py +336 -127
- package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_backends.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_bet1_spike.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claude_plugin.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_factory.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_opencode_ux.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_sweep_regressions.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_webcache.cpython-311.pyc +0 -0
- package/runner/tests/test_backends.py +280 -0
- package/runner/tests/test_sweep_regressions.py +71 -38
- package/scripts/__pycache__/bet1_advisory_spike.cpython-311.pyc +0 -0
- package/scripts/__pycache__/build_adapters.cpython-311.pyc +0 -0
- package/scripts/__pycache__/divergence_experiment.cpython-311.pyc +0 -0
- package/scripts/build_adapters.py +27 -18
- package/skills/darkharvest/scripts/harvest.py +76 -39
- package/skills/epistemic_search/scripts/__pycache__/search.cpython-311.pyc +0 -0
- package/skills/epistemic_search/scripts/search.py +29 -19
- package/skills/epistemic_search/scripts/webcache.py +29 -17
- package/skills/factory/scripts/factory.py +1 -1
- package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
- package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
- package/skills/swarm_config/configure.py +70 -37
|
@@ -208,7 +208,9 @@ def append_ledger(base_dir: Path, events: List[StatusEvent]) -> Path:
|
|
|
208
208
|
return path
|
|
209
209
|
|
|
210
210
|
|
|
211
|
-
def queue_requeue(
|
|
211
|
+
def queue_requeue(
|
|
212
|
+
base_dir: Path, scope_id: str, reason: str = "claim degradation"
|
|
213
|
+
) -> Path:
|
|
212
214
|
"""Record that a scope needs a re-run on the next swarm pass."""
|
|
213
215
|
path = Path(base_dir) / "requeue.json"
|
|
214
216
|
entries: List[Dict[str, Any]] = []
|
|
@@ -248,7 +250,52 @@ def load_requeue(base_dir: Path) -> List[Dict[str, Any]]:
|
|
|
248
250
|
# ---- one-pass staleness check (the probe's unit of work) ----------------
|
|
249
251
|
|
|
250
252
|
|
|
251
|
-
def
|
|
253
|
+
def _scope_dirs_for(scratch: Path, scope_ids: Optional[List[str]]):
|
|
254
|
+
"""Scope dirs to fold; None means the workspace has no scratchpad yet."""
|
|
255
|
+
if scope_ids is not None:
|
|
256
|
+
return [scratch / s for s in scope_ids]
|
|
257
|
+
if not scratch.exists():
|
|
258
|
+
return None
|
|
259
|
+
return sorted(p for p in scratch.iterdir() if p.is_dir())
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def _fold_scope(scope_dir: Path, retractions) -> Tuple[List[StatusEvent], int, bool]:
|
|
263
|
+
"""Fold one scope's dossiers: (events, claim_sets, degraded)."""
|
|
264
|
+
from runner.claim_witness import claims_from_dossier, load_dossier
|
|
265
|
+
|
|
266
|
+
events: List[StatusEvent] = []
|
|
267
|
+
claim_sets = 0
|
|
268
|
+
for dossier_name in ("alpha_dossier.json", "beta_dossier.json"):
|
|
269
|
+
dpath = scope_dir / dossier_name
|
|
270
|
+
if not dpath.exists():
|
|
271
|
+
continue
|
|
272
|
+
claim_sets += 1
|
|
273
|
+
claims = claims_from_dossier(load_dossier(dpath))
|
|
274
|
+
_, scope_events = apply_degradation(
|
|
275
|
+
claims, retractions, scope_id=scope_dir.name
|
|
276
|
+
)
|
|
277
|
+
events.extend(scope_events)
|
|
278
|
+
return events, claim_sets, bool(events)
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def _mirror_events(base_dir: Path, events: List[StatusEvent]) -> None:
|
|
282
|
+
"""Best-effort mirror into the derived claim store (never fatal)."""
|
|
283
|
+
try:
|
|
284
|
+
from runner.claim_store import ClaimStore
|
|
285
|
+
|
|
286
|
+
store = ClaimStore(base_dir)
|
|
287
|
+
for e in events:
|
|
288
|
+
store.record_status_event(
|
|
289
|
+
e.scope_id, e.claim_id, e.from_status, e.to_status, e.reason, e.at
|
|
290
|
+
)
|
|
291
|
+
store.close()
|
|
292
|
+
except Exception as exc: # noqa: BLE001
|
|
293
|
+
print(f"[living-dossiers] claim store mirror skipped: {exc}", file=sys.stderr)
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def check_staleness(
|
|
297
|
+
base_dir: Path, scope_ids: Optional[List[str]] = None
|
|
298
|
+
) -> Dict[str, Any]:
|
|
252
299
|
"""Run one degradation pass over scope dossiers.
|
|
253
300
|
|
|
254
301
|
For each scope with alpha/beta dossiers, fold claims, join against
|
|
@@ -258,83 +305,54 @@ def check_staleness(base_dir: Path, scope_ids: Optional[List[str]] = None) -> Di
|
|
|
258
305
|
|
|
259
306
|
Returns a summary dict suitable for JSON tool output.
|
|
260
307
|
"""
|
|
261
|
-
from runner.claim_witness import claims_from_dossier, load_dossier
|
|
262
|
-
|
|
263
308
|
base_dir = Path(base_dir)
|
|
264
309
|
retractions = load_retractions(base_dir)
|
|
265
310
|
scratch = base_dir / "scratchpads"
|
|
266
311
|
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
scope_dirs = sorted(p for p in scratch.iterdir() if p.is_dir())
|
|
278
|
-
else:
|
|
279
|
-
scope_dirs = [scratch / s for s in scope_ids]
|
|
312
|
+
scope_dirs = _scope_dirs_for(scratch, scope_ids)
|
|
313
|
+
if scope_dirs is None:
|
|
314
|
+
return {
|
|
315
|
+
"scopes": [],
|
|
316
|
+
"retractions": len(retractions),
|
|
317
|
+
"degraded": 0,
|
|
318
|
+
# Keep the early return shape identical to the normal one so
|
|
319
|
+
# callers can key off `degraded_scopes` unconditionally.
|
|
320
|
+
"degraded_scopes": [],
|
|
321
|
+
}
|
|
280
322
|
|
|
281
323
|
all_events: List[StatusEvent] = []
|
|
282
324
|
per_scope: List[Dict[str, Any]] = []
|
|
283
|
-
degraded_scopes = set()
|
|
284
325
|
|
|
285
326
|
for scope_dir in scope_dirs:
|
|
286
327
|
if not scope_dir.is_dir():
|
|
287
328
|
continue
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
claim_sets = 0
|
|
291
|
-
|
|
292
|
-
for dossier_name in ("alpha_dossier.json", "beta_dossier.json"):
|
|
293
|
-
dpath = scope_dir / dossier_name
|
|
294
|
-
if not dpath.exists():
|
|
295
|
-
continue
|
|
296
|
-
claim_sets += 1
|
|
297
|
-
dossier = load_dossier(dpath)
|
|
298
|
-
claims = claims_from_dossier(dossier)
|
|
299
|
-
_, events = apply_degradation(
|
|
300
|
-
claims, retractions, scope_id=scope_id
|
|
301
|
-
)
|
|
302
|
-
all_events.extend(events)
|
|
303
|
-
scope_event_count += len(events)
|
|
304
|
-
if events:
|
|
305
|
-
degraded_scopes.add(scope_id)
|
|
306
|
-
|
|
329
|
+
events, claim_sets, degraded = _fold_scope(scope_dir, retractions)
|
|
330
|
+
all_events.extend(events)
|
|
307
331
|
per_scope.append(
|
|
308
332
|
{
|
|
309
|
-
"scope_id":
|
|
333
|
+
"scope_id": scope_dir.name,
|
|
310
334
|
"claim_sets": claim_sets,
|
|
311
|
-
"status_events":
|
|
312
|
-
"degraded":
|
|
335
|
+
"status_events": len(events),
|
|
336
|
+
"degraded": degraded,
|
|
313
337
|
}
|
|
314
338
|
)
|
|
315
339
|
|
|
316
340
|
if all_events:
|
|
317
341
|
append_ledger(base_dir, all_events)
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
store.close()
|
|
328
|
-
except Exception as exc: # noqa: BLE001
|
|
329
|
-
print(f"[living-dossiers] claim store mirror skipped: {exc}", file=sys.stderr)
|
|
330
|
-
|
|
331
|
-
for scope_id in sorted(degraded_scopes):
|
|
332
|
-
queue_requeue(base_dir, scope_id, reason="claim degradation (retraction event)")
|
|
342
|
+
_mirror_events(base_dir, all_events)
|
|
343
|
+
|
|
344
|
+
for scope_dir in per_scope:
|
|
345
|
+
if scope_dir["degraded"]:
|
|
346
|
+
queue_requeue(
|
|
347
|
+
base_dir,
|
|
348
|
+
scope_dir["scope_id"],
|
|
349
|
+
reason="claim degradation (retraction event)",
|
|
350
|
+
)
|
|
333
351
|
|
|
334
352
|
return {
|
|
335
353
|
"retractions": len(retractions),
|
|
336
354
|
"scopes": per_scope,
|
|
337
|
-
"degraded_scopes": sorted(
|
|
355
|
+
"degraded_scopes": sorted(s["scope_id"] for s in per_scope if s["degraded"]),
|
|
338
356
|
"status_events": len(all_events),
|
|
339
357
|
}
|
|
340
358
|
|
package/runner/mcp_server.py
CHANGED
|
@@ -68,8 +68,16 @@ DEFAULT_RESEARCH_DIR = ".research"
|
|
|
68
68
|
|
|
69
69
|
|
|
70
70
|
def _resolve_base_dir(args: Dict[str, Any]) -> Path:
|
|
71
|
-
raw = args.get("base_dir") or args.get("dir")
|
|
72
|
-
|
|
71
|
+
raw = args.get("base_dir") or args.get("dir")
|
|
72
|
+
if raw:
|
|
73
|
+
return Path(os.path.realpath(str(raw)))
|
|
74
|
+
# Project root wins over process cwd: a relative ".research" resolved
|
|
75
|
+
# against the MCP server's cwd escapes the project (observed live:
|
|
76
|
+
# evidence landed in ~/.research instead of the project tree).
|
|
77
|
+
project = os.environ.get("IUMBTEMS_PROJECT_DIR", "").strip()
|
|
78
|
+
if project and Path(project).is_dir():
|
|
79
|
+
return Path(os.path.realpath(os.path.join(project, DEFAULT_RESEARCH_DIR)))
|
|
80
|
+
return Path(os.path.realpath(DEFAULT_RESEARCH_DIR))
|
|
73
81
|
|
|
74
82
|
|
|
75
83
|
def _handle_config(args: Dict[str, Any]) -> str:
|
|
@@ -123,6 +131,11 @@ def _run_swarm_mode(mode: Optional[str], args: Dict[str, Any]) -> str:
|
|
|
123
131
|
raise ValueError("objective is required")
|
|
124
132
|
|
|
125
133
|
frontier = args.get("frontier") or args.get("frontier_file")
|
|
134
|
+
overrides = None
|
|
135
|
+
backend = (args.get("backend") or "").strip().lower()
|
|
136
|
+
if backend in ("claude", "opencode"):
|
|
137
|
+
argv = ["claude", "-p"] if backend == "claude" else ["opencode", "run"]
|
|
138
|
+
overrides = {"alpha": {"backend": argv}, "beta": {"backend": argv}}
|
|
126
139
|
runner = SwarmRunner(
|
|
127
140
|
base_dir=_resolve_base_dir(args),
|
|
128
141
|
mock_mode=bool(args.get("mock_claude") or args.get("mock_mode")),
|
|
@@ -130,6 +143,7 @@ def _run_swarm_mode(mode: Optional[str], args: Dict[str, Any]) -> str:
|
|
|
130
143
|
engine=args.get("engine"),
|
|
131
144
|
depth=args.get("depth"),
|
|
132
145
|
domain_pack=args.get("domain_pack"),
|
|
146
|
+
agent_overrides=overrides,
|
|
133
147
|
)
|
|
134
148
|
buf = io.StringIO()
|
|
135
149
|
with redirect_stdout(buf):
|
|
@@ -423,6 +437,11 @@ def build_tools() -> List[ToolSpec]:
|
|
|
423
437
|
"type": "boolean",
|
|
424
438
|
"description": "Synthetic run, zero API cost",
|
|
425
439
|
},
|
|
440
|
+
"backend": {
|
|
441
|
+
"type": "string",
|
|
442
|
+
"enum": ["auto", "claude", "opencode"],
|
|
443
|
+
"description": "Agent runtime backend (default auto = host-native)",
|
|
444
|
+
},
|
|
426
445
|
},
|
|
427
446
|
"required": ["objective"],
|
|
428
447
|
},
|
|
@@ -440,6 +459,10 @@ def build_tools() -> List[ToolSpec]:
|
|
|
440
459
|
},
|
|
441
460
|
"base_dir": {"type": "string"},
|
|
442
461
|
"mock_claude": {"type": "boolean"},
|
|
462
|
+
"backend": {
|
|
463
|
+
"type": "string",
|
|
464
|
+
"enum": ["auto", "claude", "opencode"],
|
|
465
|
+
},
|
|
443
466
|
},
|
|
444
467
|
"required": ["target"],
|
|
445
468
|
},
|
|
@@ -457,6 +480,10 @@ def build_tools() -> List[ToolSpec]:
|
|
|
457
480
|
},
|
|
458
481
|
"base_dir": {"type": "string"},
|
|
459
482
|
"mock_claude": {"type": "boolean"},
|
|
483
|
+
"backend": {
|
|
484
|
+
"type": "string",
|
|
485
|
+
"enum": ["auto", "claude", "opencode"],
|
|
486
|
+
},
|
|
460
487
|
},
|
|
461
488
|
"required": ["feature"],
|
|
462
489
|
},
|
|
@@ -474,6 +501,10 @@ def build_tools() -> List[ToolSpec]:
|
|
|
474
501
|
},
|
|
475
502
|
"base_dir": {"type": "string"},
|
|
476
503
|
"mock_claude": {"type": "boolean"},
|
|
504
|
+
"backend": {
|
|
505
|
+
"type": "string",
|
|
506
|
+
"enum": ["auto", "claude", "opencode"],
|
|
507
|
+
},
|
|
477
508
|
},
|
|
478
509
|
"required": ["objective"],
|
|
479
510
|
},
|
|
@@ -501,6 +532,10 @@ def build_tools() -> List[ToolSpec]:
|
|
|
501
532
|
"depth": {"type": "integer", "description": "Dialectic depth 1-4"},
|
|
502
533
|
"base_dir": {"type": "string"},
|
|
503
534
|
"mock_claude": {"type": "boolean"},
|
|
535
|
+
"backend": {
|
|
536
|
+
"type": "string",
|
|
537
|
+
"enum": ["auto", "claude", "opencode"],
|
|
538
|
+
},
|
|
504
539
|
},
|
|
505
540
|
"required": ["objective"],
|
|
506
541
|
},
|
package/runner/pcrb.py
CHANGED
|
@@ -83,10 +83,14 @@ def _resolve_key(key: Optional[str], key_path: Optional[Path]) -> Optional[bytes
|
|
|
83
83
|
return None
|
|
84
84
|
|
|
85
85
|
|
|
86
|
-
def sign_manifest(
|
|
86
|
+
def sign_manifest(
|
|
87
|
+
manifest: Dict[str, Any], key: Optional[bytes], key_id: str = "local"
|
|
88
|
+
) -> Dict[str, Any]:
|
|
87
89
|
if not key:
|
|
88
90
|
return {"alg": "none", "key_id": None, "sig": None}
|
|
89
|
-
sig = hmac.new(
|
|
91
|
+
sig = hmac.new(
|
|
92
|
+
key, canonical_json(manifest).encode("utf-8"), hashlib.sha256
|
|
93
|
+
).hexdigest()
|
|
90
94
|
return {"alg": "hmac-sha256", "key_id": key_id, "sig": sig}
|
|
91
95
|
|
|
92
96
|
|
|
@@ -98,35 +102,47 @@ def _resolve_synthesis(base_dir: Path) -> str:
|
|
|
98
102
|
return "\n\n---\n\n".join(p.read_text(encoding="utf-8") for p in parts)
|
|
99
103
|
|
|
100
104
|
|
|
101
|
-
def
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
105
|
+
def _witness_for(c, hasher: Any, needed_hashes: set) -> Dict[str, Any]:
|
|
106
|
+
"""Build one claim's witness record and track hashes worth bundling."""
|
|
107
|
+
witness: Dict[str, Any] = {
|
|
108
|
+
"quote": c.verbatim_quote,
|
|
109
|
+
"source_hash": c.source_hash,
|
|
110
|
+
"confidence": None,
|
|
111
|
+
"verified_at": None,
|
|
112
|
+
}
|
|
113
|
+
if not c.source_hash:
|
|
114
|
+
return witness
|
|
115
|
+
if c.verbatim_quote:
|
|
116
|
+
passed, conf, _msg = hasher.verify_quote(c.source_hash, c.verbatim_quote)
|
|
117
|
+
witness["confidence"] = float(conf)
|
|
118
|
+
witness["verified_at"] = datetime.now(timezone.utc).isoformat()
|
|
119
|
+
if passed:
|
|
120
|
+
needed_hashes.add(c.source_hash)
|
|
121
|
+
else:
|
|
122
|
+
needed_hashes.add(c.source_hash)
|
|
123
|
+
return witness
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _iter_dossier_claims(scope_dirs: List[Path]):
|
|
127
|
+
"""Yield claim witnesses from every alpha/beta dossier under scopes."""
|
|
106
128
|
for scope_dir in scope_dirs:
|
|
107
129
|
for dossier_name in ("alpha_dossier.json", "beta_dossier.json"):
|
|
108
130
|
dpath = scope_dir / dossier_name
|
|
109
131
|
if not dpath.exists():
|
|
110
132
|
continue
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
if passed and c.source_hash:
|
|
125
|
-
needed_hashes.add(c.source_hash)
|
|
126
|
-
elif c.source_hash:
|
|
127
|
-
needed_hashes.add(c.source_hash)
|
|
128
|
-
rec["witness"] = witness
|
|
129
|
-
claim_records.append(rec)
|
|
133
|
+
for c in claims_from_dossier(load_dossier(dpath)):
|
|
134
|
+
yield c
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def _collect_claims(
|
|
138
|
+
scope_dirs: List[Path], hasher: Any
|
|
139
|
+
) -> Tuple[List[Dict[str, Any]], set]:
|
|
140
|
+
claim_records: List[Dict[str, Any]] = []
|
|
141
|
+
needed_hashes: set = set()
|
|
142
|
+
for c in _iter_dossier_claims(scope_dirs):
|
|
143
|
+
rec = c.to_dict()
|
|
144
|
+
rec["witness"] = _witness_for(c, hasher, needed_hashes)
|
|
145
|
+
claim_records.append(rec)
|
|
130
146
|
return claim_records, needed_hashes
|
|
131
147
|
|
|
132
148
|
|
|
@@ -203,7 +219,11 @@ def export_brief(
|
|
|
203
219
|
"signature": signature,
|
|
204
220
|
}
|
|
205
221
|
|
|
206
|
-
out =
|
|
222
|
+
out = (
|
|
223
|
+
Path(os.path.realpath(str(out_path)))
|
|
224
|
+
if out_path
|
|
225
|
+
else (base_dir / "brief.pcrb.json")
|
|
226
|
+
)
|
|
207
227
|
out.parent.mkdir(parents=True, exist_ok=True)
|
|
208
228
|
with open(out, "w", encoding="utf-8") as f:
|
|
209
229
|
json.dump(bundle, f, indent=2)
|
|
@@ -213,7 +233,9 @@ def export_brief(
|
|
|
213
233
|
def main(argv: Optional[List[str]] = None) -> int:
|
|
214
234
|
import argparse
|
|
215
235
|
|
|
216
|
-
parser = argparse.ArgumentParser(
|
|
236
|
+
parser = argparse.ArgumentParser(
|
|
237
|
+
description="Export a proof-carrying research brief"
|
|
238
|
+
)
|
|
217
239
|
parser.add_argument("--dir", default=".research")
|
|
218
240
|
parser.add_argument("--out", default=None)
|
|
219
241
|
parser.add_argument("--scope", action="append", dest="scope_ids")
|
|
@@ -224,7 +246,9 @@ def main(argv: Optional[List[str]] = None) -> int:
|
|
|
224
246
|
|
|
225
247
|
canonical_dir = Path(os.path.realpath(str(args.dir)))
|
|
226
248
|
canonical_out = Path(os.path.realpath(str(args.out))) if args.out else None
|
|
227
|
-
canonical_key =
|
|
249
|
+
canonical_key = (
|
|
250
|
+
Path(os.path.realpath(str(args.key_file))) if args.key_file else None
|
|
251
|
+
)
|
|
228
252
|
path = export_brief(
|
|
229
253
|
canonical_dir,
|
|
230
254
|
out_path=canonical_out,
|
package/runner/refinement.py
CHANGED
|
@@ -52,9 +52,7 @@ class Constitution:
|
|
|
52
52
|
"""Scoring + enforcement policy. Defaults == legacy auditor behavior."""
|
|
53
53
|
|
|
54
54
|
# Per-verified-claim weight. Flat 1.0 reproduces legacy E(D).
|
|
55
|
-
tier_weights: Dict[str, float] = field(
|
|
56
|
-
default_factory=lambda: {"__default__": 1.0}
|
|
57
|
-
)
|
|
55
|
+
tier_weights: Dict[str, float] = field(default_factory=lambda: {"__default__": 1.0})
|
|
58
56
|
neg_bonus: float = 0.5
|
|
59
57
|
reject_penalty: float = 2.5
|
|
60
58
|
accept_threshold: float = 0.65
|
|
@@ -144,9 +142,12 @@ def claim_verdict(
|
|
|
144
142
|
reasons.append(
|
|
145
143
|
f"ZERO_TOLERANCE_RETRACTION: source {claim.source_hash} was retracted"
|
|
146
144
|
)
|
|
147
|
-
if
|
|
148
|
-
|
|
149
|
-
|
|
145
|
+
if (
|
|
146
|
+
claim.status in (STATUS_STALE, STATUS_SUSPECT)
|
|
147
|
+
and constitution.retraction_policy == "zero_tolerance"
|
|
148
|
+
and not any(r.startswith("ZERO_TOLERANCE_RETRACTION") for r in reasons)
|
|
149
|
+
):
|
|
150
|
+
reasons.append(f"ZERO_TOLERANCE_RETRACTION: claim status {claim.status}")
|
|
150
151
|
|
|
151
152
|
return {
|
|
152
153
|
"claim_id": claim.claim_id,
|
|
@@ -155,33 +156,65 @@ def claim_verdict(
|
|
|
155
156
|
}
|
|
156
157
|
|
|
157
158
|
|
|
158
|
-
def
|
|
159
|
+
def _check_verified_claim(c: ClaimWitness, hasher: Any, out: List[Violation]) -> None:
|
|
160
|
+
"""Verified/claim-kind gates: hash, quote, then live witness check."""
|
|
161
|
+
if not c.source_hash:
|
|
162
|
+
out.append(
|
|
163
|
+
Violation(c.claim_id, "VERIFIED_REQUIRES_HASH", "source_hash missing")
|
|
164
|
+
)
|
|
165
|
+
if not c.verbatim_quote:
|
|
166
|
+
out.append(
|
|
167
|
+
Violation(c.claim_id, "VERIFIED_REQUIRES_QUOTE", "verbatim_quote missing")
|
|
168
|
+
)
|
|
169
|
+
if hasher is None or not (c.source_hash and c.verbatim_quote):
|
|
170
|
+
return
|
|
171
|
+
passed, _conf, msg = witness_check(c, hasher)
|
|
172
|
+
if not passed:
|
|
173
|
+
out.append(
|
|
174
|
+
Violation(c.claim_id, "WITNESS_CHECK_FAILED", msg or "quote not in source")
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _check_verified_and_inferred(
|
|
179
|
+
c: ClaimWitness, hasher: Any, out: List[Violation]
|
|
180
|
+
) -> None:
|
|
159
181
|
if c.tag == TAG_VERIFIED or c.kind == KIND_CLAIM:
|
|
160
|
-
|
|
161
|
-
out.append(Violation(c.claim_id, "VERIFIED_REQUIRES_HASH", "source_hash missing"))
|
|
162
|
-
if not c.verbatim_quote:
|
|
163
|
-
out.append(Violation(c.claim_id, "VERIFIED_REQUIRES_QUOTE", "verbatim_quote missing"))
|
|
164
|
-
if hasher is not None and c.source_hash and c.verbatim_quote:
|
|
165
|
-
passed, _conf, msg = witness_check(c, hasher)
|
|
166
|
-
if not passed:
|
|
167
|
-
out.append(Violation(c.claim_id, "WITNESS_CHECK_FAILED", msg or "quote not in source"))
|
|
182
|
+
_check_verified_claim(c, hasher, out)
|
|
168
183
|
|
|
169
184
|
if c.tag == TAG_INFERRED or c.kind == KIND_INFERENCE:
|
|
170
185
|
if not c.parent_claims:
|
|
171
|
-
out.append(
|
|
186
|
+
out.append(
|
|
187
|
+
Violation(
|
|
188
|
+
c.claim_id, "INFERRED_REQUIRES_PARENTS", "parent_claims empty"
|
|
189
|
+
)
|
|
190
|
+
)
|
|
172
191
|
if not c.deductive_logic:
|
|
173
|
-
out.append(
|
|
192
|
+
out.append(
|
|
193
|
+
Violation(
|
|
194
|
+
c.claim_id, "INFERRED_REQUIRES_LOGIC", "deductive_logic missing"
|
|
195
|
+
)
|
|
196
|
+
)
|
|
174
197
|
|
|
175
198
|
|
|
176
199
|
def _check_hypotheses_and_neg_knowledge(c: ClaimWitness, out: List[Violation]) -> None:
|
|
177
200
|
if (c.tag == TAG_HYPOTHESIS or c.kind == KIND_HYPOTHESIS) and not c.falsification:
|
|
178
|
-
out.append(
|
|
201
|
+
out.append(
|
|
202
|
+
Violation(
|
|
203
|
+
c.claim_id, "HYPOTHESIS_REQUIRES_FALSIFICATION", "falsification missing"
|
|
204
|
+
)
|
|
205
|
+
)
|
|
179
206
|
|
|
180
207
|
if c.tag == TAG_NEGATIVE_KNOWLEDGE or c.kind == KIND_NEGATIVE_KNOWLEDGE:
|
|
181
208
|
if not c.query:
|
|
182
|
-
out.append(
|
|
209
|
+
out.append(
|
|
210
|
+
Violation(c.claim_id, "NEG_KNOWLEDGE_REQUIRES_QUERY", "query missing")
|
|
211
|
+
)
|
|
183
212
|
if not c.finding:
|
|
184
|
-
out.append(
|
|
213
|
+
out.append(
|
|
214
|
+
Violation(
|
|
215
|
+
c.claim_id, "NEG_KNOWLEDGE_REQUIRES_FINDING", "finding missing"
|
|
216
|
+
)
|
|
217
|
+
)
|
|
185
218
|
|
|
186
219
|
|
|
187
220
|
def _check_constitution_and_parents(
|
|
@@ -193,7 +226,11 @@ def _check_constitution_and_parents(
|
|
|
193
226
|
for parent in c.parent_claims:
|
|
194
227
|
if parent not in known_ids:
|
|
195
228
|
out.append(
|
|
196
|
-
Violation(
|
|
229
|
+
Violation(
|
|
230
|
+
c.claim_id,
|
|
231
|
+
"PARENT_UNRESOLVED",
|
|
232
|
+
f"parent_claims '{parent}' not in dossier set",
|
|
233
|
+
)
|
|
197
234
|
)
|
|
198
235
|
|
|
199
236
|
if constitution.mandatory_tags and not c.tag:
|
|
@@ -204,7 +241,11 @@ def _check_constitution_and_parents(
|
|
|
204
241
|
for dom in constitution.banned_domains:
|
|
205
242
|
if dom.lower() in src_lower:
|
|
206
243
|
out.append(
|
|
207
|
-
Violation(
|
|
244
|
+
Violation(
|
|
245
|
+
c.claim_id,
|
|
246
|
+
"BANNED_DOMAIN",
|
|
247
|
+
f"source_url on banned domain '{dom}'",
|
|
248
|
+
)
|
|
208
249
|
)
|
|
209
250
|
|
|
210
251
|
|
|
@@ -261,9 +302,7 @@ def compute_epistemic_score(
|
|
|
261
302
|
# tier. A Domain Pack may override tier_weights; then each verified claim
|
|
262
303
|
# would need its tier looked up — that path is exercised by Stream G's
|
|
263
304
|
# `compute_epistemic_score_from_claims`, not here.
|
|
264
|
-
verified_weight = (
|
|
265
|
-
constitution.weight_for(None) * total_verified
|
|
266
|
-
)
|
|
305
|
+
verified_weight = constitution.weight_for(None) * total_verified
|
|
267
306
|
|
|
268
307
|
total_assertions = max(
|
|
269
308
|
1, total_verified + total_inferred + total_hypotheses + total_rejected
|
|
@@ -308,7 +347,8 @@ def compute_epistemic_score_from_claims(
|
|
|
308
347
|
}
|
|
309
348
|
weight_sum = 0.0
|
|
310
349
|
tier_weighted = any(
|
|
311
|
-
not math.isclose(w, 1.0, rel_tol=1e-7)
|
|
350
|
+
not math.isclose(w, 1.0, rel_tol=1e-7)
|
|
351
|
+
for w in constitution.tier_weights.values()
|
|
312
352
|
)
|
|
313
353
|
|
|
314
354
|
for c in claims:
|
|
@@ -326,7 +366,10 @@ def compute_epistemic_score_from_claims(
|
|
|
326
366
|
|
|
327
367
|
total_assertions = max(
|
|
328
368
|
1,
|
|
329
|
-
counts["verified"]
|
|
369
|
+
counts["verified"]
|
|
370
|
+
+ counts["inferred"]
|
|
371
|
+
+ counts["hypotheses"]
|
|
372
|
+
+ counts["rejected"],
|
|
330
373
|
)
|
|
331
374
|
raw_score = (
|
|
332
375
|
weight_sum
|