eduevidence 6.0.0 → 6.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (267) hide show
  1. package/CHANGELOG.md +395 -0
  2. package/CONTRIBUTING.md +105 -0
  3. package/README.md +113 -49
  4. package/README.zh-CN.md +39 -12
  5. package/SKILL.md +15 -5
  6. package/assets/readme/landing-tour.gif +0 -0
  7. package/assets/readme/studio-tour.gif +0 -0
  8. package/benchmarks/evidence-library.json +277 -1
  9. package/bin/eduevidence.js +2 -1
  10. package/docs/architecture.md +325 -46
  11. package/docs/demo-workplace-ai.md +1 -1
  12. package/docs/install-guide.md +1 -1
  13. package/docs/j-ev-experimental.md +250 -0
  14. package/docs/orchestration-role-model.md +1 -1
  15. package/docs/release-closeout/README.md +1 -1
  16. package/docs/reproducibility.md +138 -0
  17. package/docs/sciverse-api.md +125 -0
  18. package/domains/_neutral/copy/few_shots.json +21 -0
  19. package/domains/_neutral/copy/framing_lexicon.json +19 -0
  20. package/domains/_neutral/copy/module_labels.json +5 -0
  21. package/domains/_neutral/copy/module_labels_footer.json +102 -0
  22. package/domains/_neutral/copy/module_labels_modules.json +204 -0
  23. package/domains/_neutral/copy/module_labels_nav.json +126 -0
  24. package/domains/_neutral/copy/module_labels_summary.json +98 -0
  25. package/domains/_neutral/copy/module_labels_tables.json +164 -0
  26. package/domains/_neutral/copy/module_labels_v2.json +90 -0
  27. package/domains/_neutral/copy/risk_constructs.json +20 -0
  28. package/domains/_neutral/copy/section_titles.json +66 -0
  29. package/domains/_neutral/copy/terminology.json +11 -0
  30. package/domains/check_copy_packs.py +103 -0
  31. package/domains/education/copy/few_shots.json +22 -0
  32. package/domains/education/copy/framing_enums.json +167 -0
  33. package/domains/education/copy/framing_lexicon.json +166 -0
  34. package/domains/education/copy/module_labels.json +169 -0
  35. package/domains/education/copy/risk_constructs.json +48 -0
  36. package/domains/education/copy/section_titles.json +186 -0
  37. package/domains/education/copy/terminology.json +70 -0
  38. package/domains/education/manifest.json +1 -1
  39. package/domains/education/outcome_taxonomy.json +2 -2
  40. package/domains/manifest.json +1 -1
  41. package/domains/policy/copy/few_shots.json +22 -0
  42. package/domains/policy/copy/framing_enums.json +94 -0
  43. package/domains/policy/copy/framing_lexicon.json +174 -0
  44. package/domains/policy/copy/module_labels.json +168 -0
  45. package/domains/policy/copy/risk_constructs.json +33 -0
  46. package/domains/policy/copy/section_titles.json +186 -0
  47. package/domains/policy/copy/terminology.json +64 -0
  48. package/eduevidence_cli.py +10 -0
  49. package/engine/capabilities.py +57 -5
  50. package/engine/decision_policy.py +167 -0
  51. package/engine/evidence_graph.py +14 -10
  52. package/engine/gaps.py +42 -22
  53. package/engine/ids.py +2 -0
  54. package/engine/library.py +6 -2
  55. package/engine/library_builtin.py +7 -4
  56. package/engine/living.py +34 -4
  57. package/engine/migration.py +88 -3
  58. package/engine/orchestration.py +5 -5
  59. package/engine/paths.py +2 -0
  60. package/engine/pilot.py +34 -32
  61. package/engine/taxonomy.py +211 -0
  62. package/engine/tribunal.py +49 -43
  63. package/engine/versions.py +1 -1
  64. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1361 -147
  65. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +3 -3
  66. package/examples/ai-coding-assistant-evidence/citation_check.json +1 -1
  67. package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
  68. package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
  69. package/examples/ai-coding-assistant-evidence/report_spec.json +23 -12
  70. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +448 -128
  71. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +448 -128
  72. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +448 -128
  73. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +448 -128
  74. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +448 -128
  75. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1360 -146
  76. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1360 -146
  77. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1360 -146
  78. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1360 -146
  79. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1360 -146
  80. package/examples/ai-coding-assistant-evidence/result.json +13 -9
  81. package/examples/ai-coding-assistant-evidence/result.zh.json +45 -41
  82. package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
  83. package/examples/ai-coding-assistant-evidence/verdict.json +6 -2
  84. package/examples/spaced-retrieval-practice/EduEvidence_Report.html +2728 -0
  85. package/examples/spaced-retrieval-practice/applicability.json +14 -0
  86. package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
  87. package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
  88. package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
  89. package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
  90. package/examples/spaced-retrieval-practice/frame.json +58 -0
  91. package/examples/spaced-retrieval-practice/gate_report.json +101 -0
  92. package/examples/spaced-retrieval-practice/methodology.json +78 -0
  93. package/examples/spaced-retrieval-practice/report.html +2522 -0
  94. package/examples/spaced-retrieval-practice/report_spec.json +212 -0
  95. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
  96. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
  97. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
  98. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
  99. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
  100. package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
  101. package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
  102. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
  103. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
  104. package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
  105. package/examples/spaced-retrieval-practice/result.json +942 -0
  106. package/examples/spaced-retrieval-practice/result.zh.json +942 -0
  107. package/examples/spaced-retrieval-practice/skeptic.json +70 -0
  108. package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
  109. package/examples/spaced-retrieval-practice/verdict.json +93 -0
  110. package/examples/workplace-ai-assistant/EduEvidence_Report.html +2814 -0
  111. package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
  112. package/examples/workplace-ai-assistant/claims.jsonl +4 -4
  113. package/examples/workplace-ai-assistant/evidence.jsonl +4 -4
  114. package/examples/workplace-ai-assistant/evidence_graph.json +15 -15
  115. package/examples/workplace-ai-assistant/final_verdict.json +78 -0
  116. package/examples/workplace-ai-assistant/gate_report.json +101 -0
  117. package/examples/workplace-ai-assistant/report_spec.json +209 -40
  118. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +449 -119
  119. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +449 -119
  120. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +449 -119
  121. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +449 -119
  122. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +449 -119
  123. package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
  124. package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
  125. package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
  126. package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
  127. package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
  128. package/examples/workplace-ai-assistant/result.json +82 -20
  129. package/examples/workplace-ai-assistant/result.zh.json +82 -20
  130. package/examples/workplace-ai-assistant/skeptic.json +72 -0
  131. package/examples/workplace-ai-assistant/verdict.json +36 -10
  132. package/integrations/agent_mcp.py +2 -2
  133. package/integrations/jev/__init__.py +115 -0
  134. package/integrations/jev/approval.py +212 -0
  135. package/integrations/jev/cli.py +84 -0
  136. package/integrations/jev/config.py +112 -0
  137. package/integrations/jev/gateway.py +128 -0
  138. package/integrations/jev/modes.py +38 -0
  139. package/integrations/jev/tools_classify.py +88 -0
  140. package/integrations/jev/tools_extract.py +111 -0
  141. package/integrations/jev/tools_rerank.py +71 -0
  142. package/integrations/jev/tools_screen.py +87 -0
  143. package/integrations/jev/tools_verify.py +95 -0
  144. package/integrations/jev_mcp.py +22 -0
  145. package/integrations/semantic_decide.py +286 -0
  146. package/integrations/semdecide_cli.py +55 -0
  147. package/package.json +19 -2
  148. package/pyproject.toml +4 -3
  149. package/references/report-copy-style.md +107 -0
  150. package/references/retrieval-compliance.md +75 -0
  151. package/references/retrieval-protocol.md +20 -0
  152. package/retrieval/audit.py +27 -3
  153. package/retrieval/fetch.py +96 -0
  154. package/retrieval/sciverse.py +398 -0
  155. package/retrieval/search.py +47 -7
  156. package/schemas/applicability.schema.json +94 -0
  157. package/schemas/chart-spec.schema.json +10 -3
  158. package/schemas/evidence.schema.json +316 -43
  159. package/schemas/fetch-result.schema.json +2 -1
  160. package/schemas/report-result.schema.json +3 -3
  161. package/schemas/report-spec.schema.json +98 -100
  162. package/schemas/skeptic.schema.json +86 -0
  163. package/schemas/source.schema.json +21 -2
  164. package/schemas/v2/decision-snapshot.schema.json +20 -9
  165. package/schemas/v2/finding.schema.json +5 -1
  166. package/schemas/v2/intake.schema.json +191 -0
  167. package/schemas/v2/methodology-audit.schema.json +5 -1
  168. package/schemas/v2/outcome.schema.json +28 -5
  169. package/schemas/v2/study.schema.json +5 -1
  170. package/schemas/vNext/autoevolve-session.schema.json +34 -1
  171. package/schemas/vNext/eval-snapshot.schema.json +77 -1
  172. package/schemas/vNext/execution-plan.schema.json +50 -1
  173. package/schemas/vNext/gap-priority.schema.json +54 -1
  174. package/schemas/vNext/negative-search-record.schema.json +68 -1
  175. package/schemas/vNext/research-iteration.schema.json +87 -1
  176. package/schemas/vNext/research-strategy.schema.json +62 -1
  177. package/schemas/vNext/skill-experiment.schema.json +90 -1
  178. package/schemas/vNext/task-spec.schema.json +156 -1
  179. package/schemas/vNext/worker-result.schema.json +60 -1
  180. package/schemas/verdict.schema.json +164 -28
  181. package/scripts/build_evidence_library.py +15 -5
  182. package/scripts/build_report_variants.py +18 -2
  183. package/scripts/build_result.py +74 -9
  184. package/scripts/check_package_parity.py +85 -0
  185. package/scripts/check_protocol_alignment.py +375 -0
  186. package/scripts/check_versioned_schemas.py +254 -0
  187. package/scripts/claim_audit.py +13 -8
  188. package/scripts/compute_confidence.py +10 -0
  189. package/scripts/dashboard_server.py +13 -2
  190. package/scripts/did_regression.py +12 -2
  191. package/scripts/evidence_score.py +5 -2
  192. package/scripts/intake/__init__.py +31 -0
  193. package/scripts/intake/__main__.py +18 -0
  194. package/scripts/intake/background.py +78 -0
  195. package/scripts/intake/browser.py +79 -0
  196. package/scripts/intake/cli.py +57 -0
  197. package/scripts/intake/constants.py +57 -0
  198. package/scripts/intake/depth.py +53 -0
  199. package/scripts/intake/enhancements.py +106 -0
  200. package/scripts/intake/hooks.py +90 -0
  201. package/scripts/intake/prefs.py +76 -0
  202. package/scripts/intake/prompts.py +85 -0
  203. package/scripts/intake/session.py +152 -0
  204. package/scripts/lint_file_layers.py +126 -0
  205. package/scripts/orchestrator.py +187 -40
  206. package/scripts/pre_verdict_gate.py +241 -29
  207. package/scripts/quickstart.py +18 -2
  208. package/scripts/run_workspace.py +7 -1
  209. package/scripts/skill_lint.py +11 -1
  210. package/scripts/skill_payload.py +6 -3
  211. package/scripts/test_adversarial_empirical.py +96 -25
  212. package/scripts/validate_schema.py +31 -1
  213. package/skill/agents/evaluation-designer.md +20 -4
  214. package/skill/agents/evidence-analyst.md +19 -3
  215. package/skill/agents/evidence-judge.md +98 -8
  216. package/skill/agents/evidence-retriever.md +20 -3
  217. package/skill/agents/intervention-designer.md +20 -4
  218. package/skill/agents/method-reviewer.md +18 -2
  219. package/skill/agents/{education-planner.md → research-planner.md} +19 -3
  220. package/skill/agents/skeptic.md +18 -2
  221. package/skill/roles/registry.yaml +11 -11
  222. package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
  223. package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
  224. package/skill/sub-skills/data-analysis/SKILL.md +34 -15
  225. package/skill/sub-skills/ethics-review/SKILL.md +33 -10
  226. package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
  227. package/skill/sub-skills/evidence-review/SKILL.md +31 -12
  228. package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
  229. package/skill/sub-skills/literature-review/SKILL.md +35 -14
  230. package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
  231. package/skill/sub-skills/report-generation/SKILL.md +28 -0
  232. package/skill/sub-skills/research-planning/SKILL.md +41 -14
  233. package/skill/sub-skills/study-design/SKILL.md +30 -9
  234. package/skill/task-briefs/adjudicate.md +32 -7
  235. package/skill/task-briefs/applicability.md +37 -2
  236. package/skill/task-briefs/audit.md +32 -7
  237. package/skill/task-briefs/challenge.md +34 -5
  238. package/skill/task-briefs/evaluate.md +30 -5
  239. package/skill/task-briefs/extract.md +31 -8
  240. package/skill/task-briefs/frame.md +39 -10
  241. package/skill/task-briefs/intervene.md +32 -6
  242. package/skill/task-briefs/present.md +32 -8
  243. package/skill/task-briefs/projection.md +36 -2
  244. package/skill/task-briefs/retrieve.md +36 -6
  245. package/skill/workflows/decision-and-pilot.md +76 -1
  246. package/skill/workflows/evaluate-and-update.md +83 -0
  247. package/skill/workflows/evidence-review.md +104 -0
  248. package/skill/workflows/experimental-jev.md +170 -0
  249. package/skill/workflows/intake.md +120 -0
  250. package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
  251. package/visualization/eduevidence-report/scripts/build_infographics.py +37 -15
  252. package/visualization/eduevidence-report/scripts/build_report.py +435 -575
  253. package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
  254. package/visualization/eduevidence-report/scripts/lieflat_engine.py +349 -38
  255. package/visualization/eduevidence-report/scripts/report_copy_pack.py +296 -0
  256. package/visualization/eduevidence-report/scripts/report_copy_policy_guard.py +47 -0
  257. package/visualization/eduevidence-report/scripts/zh_labels.py +141 -1
  258. package/web/architecture.html +14885 -0
  259. package/web/studio/assets/index-B8tkF44Q.css +1 -0
  260. package/web/studio/index.html +2 -2
  261. package/scripts/build_esl_artifacts.py +0 -1921
  262. package/scripts/build_killer_demo.py +0 -295
  263. package/scripts/enrich_projects_human_and_lieflat.py +0 -315
  264. package/scripts/generate_new_projects.py +0 -686
  265. package/scripts/sync_killer_demo_report.py +0 -270
  266. package/web/studio/assets/index-CzXocaGv.css +0 -1
  267. /package/web/studio/assets/{index-pa7jD7n4.js → index-CQ6Keoyc.js} +0 -0
@@ -0,0 +1,126 @@
1
+ #!/usr/bin/env python3
2
+ """scripts/lint_file_layers.py — file-layer hard gate.
3
+
4
+ Scans engine/ scripts/ integrations/ skill/ visualization/eduevidence-report/scripts/
5
+ for *.py *.js *.ts and enforces:
6
+
7
+ * single file > 300 lines → ERROR (monument / legacy → WARNING only)
8
+ * new / moved files must be ≤ 300 lines (not eligible for whitelists)
9
+ * new / moved files must be ≤ 300 lines (not eligible for the whitelist)
10
+
11
+ Monument whitelist = build_report.py, orchestrator.py, agent_mcp.py,
12
+ pre_verdict_gate.py, living.py, evidence_graph.py. Anything else over the
13
+ budget is an ERROR unless monument/legacy WARNING. Exit 1 on any ERROR.
14
+
15
+ Usage:
16
+ python3 scripts/lint_file_layers.py
17
+ Exit code 1 when any ERROR.
18
+ """
19
+ from __future__ import annotations
20
+
21
+ import sys
22
+ from pathlib import Path
23
+
24
+ ROOT = Path(__file__).resolve().parent.parent
25
+
26
+ MAX_LINES = 300
27
+
28
+ SCAN_ROOTS = (
29
+ "engine",
30
+ "scripts",
31
+ "integrations",
32
+ "skill",
33
+ "visualization/eduevidence-report/scripts",
34
+ )
35
+
36
+ # Historical monument whitelist — WARNING only (not ERROR).
37
+ # ONLY these six pre-split giants are grandfathered. New or moved files
38
+ # MUST NOT be added here — keep them ≤ MAX_LINES instead.
39
+ MONUMENT_BASENAMES = {
40
+ "build_report.py",
41
+ "orchestrator.py",
42
+ "agent_mcp.py",
43
+ "pre_verdict_gate.py",
44
+ "living.py",
45
+ "evidence_graph.py",
46
+ }
47
+
48
+ # Pre-existing large files (before the layering gate). WARNING only.
49
+ # New/moved files must never be added here — keep them ≤ MAX_LINES.
50
+ LEGACY_LARGE_BASENAMES = {
51
+ "analysis.py", "core.py", "runner.py", "graph_store.py", "library_builtin.py",
52
+ "meta_analysis.py", "migration.py", "orchestration.py", "pilot.py",
53
+ "studio_read_model.py", "tribunal.py", "benchmark_evaluator.py",
54
+ "benchmark_judge.py", "benchmark_v2.py", "benchmark_v3.py",
55
+ "build_esl_artifacts.py", "build_evidence_library.py", "build_result.py",
56
+ "check_protocol_alignment.py", "dashboard_server.py",
57
+ "enrich_projects_human_and_lieflat.py", "generate_new_projects.py",
58
+ "render_report_html.py", "research_auto_cli.py", "run_workspace.py",
59
+ "test_adversarial_empirical.py", "build_figures.py", "charts_data.py",
60
+ "lieflat_engine.py", "zh_labels.py", "library.py",
61
+ }
62
+
63
+ EXTS = {".py", ".js", ".ts"}
64
+ SKIP_DIRS = {"__pycache__", "node_modules", ".git", ".venv", "venv"}
65
+
66
+
67
+ def _iter_sources() -> list[Path]:
68
+ found: list[Path] = []
69
+ for root_name in SCAN_ROOTS:
70
+ root = ROOT / root_name
71
+ if not root.is_dir():
72
+ continue
73
+ for path in sorted(root.rglob("*")):
74
+ if not path.is_file() or path.suffix not in EXTS:
75
+ continue
76
+ if any(part in SKIP_DIRS for part in path.parts):
77
+ continue
78
+ found.append(path)
79
+ return found
80
+
81
+
82
+ def check_file_layers() -> tuple[list[str], list[str]]:
83
+ """Return (errors, warnings)."""
84
+ errors: list[str] = []
85
+ warnings: list[str] = []
86
+ for path in _iter_sources():
87
+ rel = path.relative_to(ROOT).as_posix()
88
+ try:
89
+ n = sum(1 for _ in path.open(encoding="utf-8", errors="replace"))
90
+ except OSError as exc:
91
+ errors.append(f"{rel}: unreadable ({exc})")
92
+ continue
93
+ if n <= MAX_LINES:
94
+ continue
95
+ msg = f"{rel}: {n} lines > {MAX_LINES}"
96
+ if path.name in MONUMENT_BASENAMES:
97
+ warnings.append(
98
+ f"{msg} (monument whitelist — WARNING only; split when touched)")
99
+ elif path.name in LEGACY_LARGE_BASENAMES:
100
+ warnings.append(
101
+ f"{msg} (legacy large file — WARNING only; split when touched)")
102
+ else:
103
+ errors.append(
104
+ f"{msg} (not on monument whitelist; new/moved must be ≤ {MAX_LINES})")
105
+ return errors, warnings
106
+
107
+
108
+ def main() -> int:
109
+ print("[*] Running file-layer hard gate "
110
+ f"(max {MAX_LINES} lines; monuments+legacy WARNING only: "
111
+ f"{', '.join(sorted(MONUMENT_BASENAMES))})...")
112
+ errors, warnings = check_file_layers()
113
+ for w in warnings:
114
+ print(f" ! {w}")
115
+ if errors:
116
+ print(f"[-] File-layer lint FAILED with {len(errors)} error(s):",
117
+ file=sys.stderr)
118
+ for e in errors:
119
+ print(f" • {e}", file=sys.stderr)
120
+ return 1
121
+ print(f"[+] File-layer lint PASSED: {len(warnings)} monument warning(s), 0 error(s).")
122
+ return 0
123
+
124
+
125
+ if __name__ == "__main__":
126
+ raise SystemExit(main())
@@ -60,6 +60,13 @@ from pre_verdict_gate import apply_enforcement, evaluate_workspace # noqa: E402
60
60
  from engine.versions import ENGINE_VERSION # noqa: E402
61
61
  from engine.log import enable_console_logging, get_log # noqa: E402
62
62
 
63
+ # One-shot two-round intake + prefs (scripts/intake/). Soft-import so the
64
+ # orchestrator still starts if intake is stripped from a minimal package.
65
+ try:
66
+ from intake import hooks as _intake_hooks # noqa: E402
67
+ except Exception: # pragma: no cover - optional adapter
68
+ _intake_hooks = None
69
+
63
70
  log = get_log("orchestrator")
64
71
 
65
72
  DEPTH_ALIASES = {"quick": "S", "standard": "M", "deep": "L"}
@@ -67,13 +74,17 @@ DEPTHS = ("S", "M", "L")
67
74
 
68
75
  #: Stage -> primary artifact + schema gate + whether it is locally executable.
69
76
  STAGE_SPEC: dict[str, dict[str, Any]] = {
70
- "frame": {"artifact": "frame.json", "schema": "education-frame.schema.json", "jsonl": False, "local": False},
77
+ # The frame contract belongs to the run's domain; the placeholder is
78
+ # resolved by frame_schema_for() below (education-frame.schema.json
79
+ # for education, domains/policy/frame.schema.json for policy, and so
80
+ # on for any domain registered under domains/).
81
+ "frame": {"artifact": "frame.json", "schema": "@domain_frame", "jsonl": False, "local": False},
71
82
  "retrieve": {"artifact": "sources.jsonl", "schema": "source.schema.json", "jsonl": True, "local": False},
72
83
  "extract": {"artifact": "evidence.jsonl", "schema": "evidence.schema.json", "jsonl": True, "local": False},
73
- "challenge": {"artifact": "skeptic.json", "schema": None, "jsonl": False, "local": False},
84
+ "challenge": {"artifact": "skeptic.json", "schema": "skeptic.schema.json", "jsonl": False, "local": False},
74
85
  "audit": {"artifact": "methodology.json", "schema": "methodology.schema.json", "jsonl": False, "local": False},
75
86
  "adjudicate": {"artifact": "final_verdict.json", "schema": "verdict.schema.json", "jsonl": False, "local": True},
76
- "applicability": {"artifact": "applicability.json", "schema": None, "jsonl": False, "local": False},
87
+ "applicability": {"artifact": "applicability.json", "schema": "applicability.schema.json", "jsonl": False, "local": False},
77
88
  "intervene": {"artifact": "intervention.json", "schema": "intervention.schema.json", "jsonl": False, "local": False},
78
89
  "evaluate": {"artifact": "evaluation.json", "schema": "evaluation.schema.json", "jsonl": False, "local": False},
79
90
  "projection": {"artifact": "result.json", "schema": "report-result.schema.json", "jsonl": False, "local": True},
@@ -187,11 +198,15 @@ def init_run(
187
198
  approve_agent_mcp: bool = False,
188
199
  scp_available: bool | None = None,
189
200
  approval_record: dict | None = None,
201
+ domain: str = "education",
190
202
  ) -> RunWorkspace:
191
203
  """Create the run workspace + manifest + planning artifacts (Phase 11-13)."""
192
204
  depth = DEPTH_ALIASES.get(depth, depth)
193
205
  if depth not in DEPTHS:
194
206
  raise ValueError(f"unknown depth {depth!r}; use quick/standard/deep or S/M/L")
207
+ # Validate the domain up front: every later stage gate reads it from the
208
+ # manifest, so an unknown id must fail here rather than mid-run.
209
+ domain_frame_schema(domain)
195
210
 
196
211
  try:
197
212
  from integrations.agent_mcp import detect_agent_mcp
@@ -214,6 +229,7 @@ def init_run(
214
229
  agent_mcp_available=agent_available,
215
230
  agent_mcp_approved=approve_agent_mcp,
216
231
  root=ROOT,
232
+ domain=domain,
217
233
  )
218
234
  ws.save_manifest(manifest)
219
235
 
@@ -264,7 +280,7 @@ def init_run(
264
280
  "run_id": run_id,
265
281
  "execution_mode": agent_mode,
266
282
  "routing": {
267
- "education-planner": "strong/reasoning",
283
+ "research-planner": "strong/reasoning",
268
284
  "evidence-retriever": "fast/low-cost",
269
285
  "evidence-analyst": "strong/structured",
270
286
  "skeptic": "independent/reasoning",
@@ -319,11 +335,45 @@ def _load_artifact(ws: RunWorkspace, artifact: str) -> list[dict[str, Any]]:
319
335
  return [data] if data else []
320
336
 
321
337
 
338
+
339
+ #: Placeholder replaced by the run's registered frame schema.
340
+ DOMAIN_FRAME_SENTINEL = "@domain_frame"
341
+
342
+
343
+ def domain_frame_schema(domain: str) -> str:
344
+ """Registered frame schema for a domain, relative to the repository root.
345
+
346
+ Raises ValueError for an unknown domain so a typo fails loudly instead of
347
+ silently validating against the education contract.
348
+ """
349
+ try:
350
+ from engine.evidencecore import load_domain
351
+
352
+ entry = load_domain(domain)
353
+ except KeyError as exc:
354
+ raise ValueError(str(exc)) from exc
355
+ return str(entry["frame_schema"]) # type: ignore[return-value]
356
+
357
+
358
+ def frame_schema_for(ws: "RunWorkspace") -> str:
359
+ """The frame schema this run must satisfy (from its manifest domain)."""
360
+ domain = str(ws.load_manifest().get("domain") or "education")
361
+ return domain_frame_schema(domain)
362
+
363
+
322
364
  def schema_gate(ws: RunWorkspace, stage: str) -> dict[str, Any]:
323
365
  """Validate a stage's primary artifact against its schema. Never raises."""
324
366
  spec = STAGE_SPEC[stage]
325
367
  artifact = spec["artifact"]
326
368
  schema_name = spec["schema"]
369
+ if schema_name == DOMAIN_FRAME_SENTINEL:
370
+ # The frame contract is per-domain; resolve it from the run manifest.
371
+ try:
372
+ schema_name = frame_schema_for(ws)
373
+ except ValueError as exc:
374
+ return {"passed": False, "stage": stage, "artifact": artifact,
375
+ "schema": DOMAIN_FRAME_SENTINEL,
376
+ "issues": [f"unknown run domain: {exc}"]}
327
377
  if schema_name is None: # lightweight parseability contract
328
378
  data = load_json(ws.path / artifact)
329
379
  ok = bool(data) and isinstance(data, dict)
@@ -333,13 +383,19 @@ def schema_gate(ws: RunWorkspace, stage: str) -> dict[str, Any]:
333
383
  from validate_schema import SchemaError, Validator
334
384
 
335
385
  schemas_dir = ROOT / "schemas"
336
- if not (schemas_dir / schema_name).is_file():
337
- share_dir = Path(sys.prefix) / "share" / "eduevidence" / "schemas"
338
- if (share_dir / schema_name).is_file():
339
- schemas_dir = share_dir
386
+ schema_path = schemas_dir / schema_name
387
+ if not schema_path.is_file():
388
+ # A domain may own its frame schema outside schemas/ (policy does).
389
+ candidate = ROOT / schema_name
390
+ if candidate.is_file():
391
+ schema_path = candidate
392
+ else:
393
+ share_candidates = (Path(sys.prefix) / "share" / "eduevidence" / schema_name,
394
+ Path(sys.prefix) / "share" / "eduevidence" / "schemas" / schema_name)
395
+ schema_path = next((c for c in share_candidates if c.is_file()), schema_path)
340
396
 
341
397
  try:
342
- schema = json.loads((schemas_dir / schema_name).read_text(encoding="utf-8"))
398
+ schema = json.loads(schema_path.read_text(encoding="utf-8"))
343
399
  except OSError:
344
400
  return {"passed": False, "stage": stage, "artifact": artifact,
345
401
  "schema": schema_name, "issues": [f"schema file {schema_name} not found"]}
@@ -398,22 +454,52 @@ def derive_sources_from_evidence(evidence: list[dict[str, Any]]) -> list[dict[st
398
454
  return list(seen.values())
399
455
 
400
456
 
457
+ #: The nine checks the skeptic contract requires (skill/task-briefs/challenge.md).
458
+ SKEPTIC_CHECKS = (
459
+ "1_null_result", "2_negative_result", "3_contradictory_evidence",
460
+ "4_alternative_explanation", "5_measurement_mismatch", "6_sampling_bias",
461
+ "7_novelty_effect", "8_ai_dependency", "9_scope_overreach",
462
+ )
463
+
464
+
401
465
  def derive_skeptic_from_evidence(evidence: list[dict[str, Any]]) -> dict[str, Any]:
402
466
  """Deterministic skeptic summary derived from evidence directions (demo/test mode).
403
467
 
404
468
  Records what the corpus itself contains (contradictions / null results /
405
- confounders); it never invents counter-evidence.
469
+ confounders); it never invents counter-evidence. Field names follow the
470
+ challenge brief and the skeptic role prompt so the Pre-Verdict Gate reads
471
+ the same keys this writes.
406
472
  """
407
- contradictions = [e.get("evidence_id") for e in evidence if e.get("direction") == "contradict"]
408
- null_results = [e.get("evidence_id") for e in evidence if e.get("direction") == "neutral"]
473
+ contradictions = [e.get("evidence_id") for e in evidence
474
+ if (e.get("relation_to_claim") or e.get("direction")) == "contradict"]
475
+ null_results = [e.get("evidence_id") for e in evidence
476
+ if e.get("effect_direction") == "null"]
409
477
  confounders = sorted({c for e in evidence for c in (e.get("confounders", []) or [])})
478
+
479
+ status_for = {
480
+ "1_null_result": "found" if null_results else "not_found",
481
+ "2_negative_result": "found" if any(
482
+ e.get("effect_direction") == "negative" for e in evidence) else "not_found",
483
+ "3_contradictory_evidence": "found" if contradictions else "not_found",
484
+ "4_alternative_explanation": "found" if confounders else "not_found",
485
+ }
486
+ findings = []
487
+ for check in SKEPTIC_CHECKS:
488
+ findings.append({
489
+ "check": check,
490
+ "status": status_for.get(check, "not_found"),
491
+ "detail": "derived from the evidence corpus in demo/test mode",
492
+ "related_evidence_ids": (contradictions if "contradict" in check
493
+ else null_results if "null" in check else []),
494
+ })
410
495
  return {
411
496
  "search_performed": True,
412
497
  "method": "derived from evidence corpus directions (demo/test mode)",
413
- "contradictions": contradictions,
414
- "null_results": null_results,
415
- "confounders": confounders,
416
- "no_contradictory_evidence_found": not contradictions,
498
+ "skeptic_findings": findings,
499
+ "contradictory_evidence_found": bool(contradictions),
500
+ "no_contradictory_evidence_statement": (
501
+ "" if contradictions else "NO CONTRADICTORY EVIDENCE FOUND"),
502
+ "threats_to_validity": confounders,
417
503
  }
418
504
 
419
505
 
@@ -422,7 +508,13 @@ def _cap_verdict(gate: dict[str, Any], raw_verdict: dict[str, Any],
422
508
  """Build final_verdict.json: deterministic confidence + gate enforcement."""
423
509
  final = copy.deepcopy(raw_verdict)
424
510
  final["raw_model_confidence"] = raw_verdict.get("confidence")
425
- final["raw_model_confidence_breakdown"] = raw_verdict.get("confidence_breakdown")
511
+ # The schema types this as an object; a model verdict without the field
512
+ # produced None here, which made the written file schema-invalid and the
513
+ # gate fail on data the pipeline had just produced.
514
+ final["raw_model_confidence_breakdown"] = (
515
+ raw_verdict.get("confidence_breakdown")
516
+ if isinstance(raw_verdict.get("confidence_breakdown"), dict)
517
+ else {})
426
518
  final["confidence"] = computed["confidence"]
427
519
  final["confidence_score"] = computed["confidence_breakdown"].get("score")
428
520
  final["confidence_policy_version"] = computed["confidence_policy_version"]
@@ -461,6 +553,10 @@ def _run_adjudicate(ws: RunWorkspace, question: str,
461
553
  json.dumps(final, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
462
554
 
463
555
  post = evaluate_workspace(ws.path, require_final=True)
556
+ if not post.get("passed") or post.get("critical_failures"):
557
+ final = apply_enforcement(final, post)
558
+ (ws.path / "final_verdict.json").write_text(
559
+ json.dumps(final, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
464
560
  gate_report = {"run_id": ws.run_id, "stage": "adjudicate",
465
561
  "pre": {"passed": pre["passed"], "max_confidence": pre["max_confidence"],
466
562
  "critical_failures": pre["critical_failures"]},
@@ -486,8 +582,7 @@ def _run_adjudicate(ws: RunWorkspace, question: str,
486
582
 
487
583
  def _assemble_result(ws: RunWorkspace, manifest: dict[str, Any]) -> dict[str, Any]:
488
584
  """Assemble result.json from workspace artifacts (decision=final_verdict.json)."""
489
- from build_result import (NOT_CAPTURED_USAGE, OUTCOME_ORDER,
490
- aggregate_outcomes, build_claims,
585
+ from build_result import (NOT_CAPTURED_USAGE, aggregate_outcomes, build_claims,
491
586
  build_outcome_mapping, derive_provenance)
492
587
 
493
588
  frame = load_json(ws.path / "frame.json")
@@ -530,7 +625,8 @@ def _assemble_result(ws: RunWorkspace, manifest: dict[str, Any]) -> dict[str, An
530
625
  "methodology_reviews": methodology_list,
531
626
  "conflicts": [{"reason_for_disagreement": verdict.get("reason_for_disagreement", "")}]
532
627
  if verdict.get("reason_for_disagreement") else [],
533
- "applicability": verdict.get("applicability", {}),
628
+ "applicability": (load_json(ws.path / "applicability.json")
629
+ or verdict.get("applicability", {})),
534
630
  "intervention": intervention,
535
631
  "evaluation": evaluation,
536
632
  "benchmark": {},
@@ -696,10 +792,16 @@ def _seed_from_demo(ws: RunWorkspace, stage: str, demo_pack: Path) -> dict[str,
696
792
  value = verdict.get("applicability") if isinstance(verdict, dict) else None
697
793
  # A demo can only carry the decision's existing applicability
698
794
  # boundary; absence remains explicit rather than inferred.
699
- payload = value if isinstance(value, dict) and value else {
700
- "status": "NOT_CAPTURED",
701
- "reason": "demo pack does not provide an applicability assessment",
702
- }
795
+ # The derived boundary is tagged ASSESSED so it satisfies the
796
+ # applicability contract instead of being an unlabelled dict.
797
+ if isinstance(value, dict) and value:
798
+ payload = {"status": "ASSESSED"}
799
+ payload.update(value)
800
+ else:
801
+ payload = {
802
+ "status": "NOT_CAPTURED",
803
+ "reason": "demo pack does not provide an applicability assessment",
804
+ }
703
805
  (ws.path / "applicability.json").write_text(
704
806
  json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
705
807
  return {"seeded": True, "detail": "applicability.json seeded from decision boundary (demo)"}
@@ -884,12 +986,45 @@ def load_global_approval() -> dict | None:
884
986
 
885
987
 
886
988
  def _cmd_run(args: argparse.Namespace) -> int:
887
- approve = args.approve_agent_mcp or interactive_agent_mcp_setup(args.approve_agent_mcp)
888
- approval_record = None
889
- if approve:
890
- approval_record = load_global_approval()
891
- ws = init_run(Path(args.runs_dir), args.question, depth=args.depth, run_id=args.run_id,
892
- approve_agent_mcp=approve, approval_record=approval_record)
989
+ """Create a run workspace. Delegates intake/open to scripts/intake/hooks.py."""
990
+ assume_yes = bool(getattr(args, "yes", False))
991
+ domain = getattr(args, "domain", "education")
992
+ enhancements = list(getattr(args, "enhancement", None) or []) or None
993
+ intake_record: dict[str, Any] | None = None
994
+
995
+ if _intake_hooks is not None:
996
+ ctx = _intake_hooks.prepare_run(
997
+ question=args.question,
998
+ depth=getattr(args, "depth", None),
999
+ enhancements=enhancements,
1000
+ domain=domain,
1001
+ assume_yes=assume_yes,
1002
+ )
1003
+ if ctx.get("error"):
1004
+ print(f"ERROR: intake failed: {ctx['error']}", file=sys.stderr)
1005
+ return 2
1006
+ question, depth_arg = ctx["question"], ctx["depth"]
1007
+ enhancements, intake_record = ctx["enhancements"], ctx["intake_record"]
1008
+ wants_mcp = ctx["wants_mcp"]
1009
+ else:
1010
+ question = args.question
1011
+ depth_arg = DEPTH_ALIASES.get(getattr(args, "depth", None) or "M",
1012
+ getattr(args, "depth", None) or "M")
1013
+ if depth_arg not in DEPTHS:
1014
+ depth_arg = "M"
1015
+ wants_mcp = bool(enhancements and "agent_mcp" in enhancements
1016
+ and "none" not in enhancements)
1017
+
1018
+ approve = bool(args.approve_agent_mcp)
1019
+ if wants_mcp and not assume_yes and sys.stdin.isatty():
1020
+ approve = approve or interactive_agent_mcp_setup(approve)
1021
+ approval_record = load_global_approval() if approve else None
1022
+
1023
+ ws = init_run(Path(args.runs_dir), question, depth=depth_arg, run_id=args.run_id,
1024
+ approve_agent_mcp=approve, approval_record=approval_record,
1025
+ domain=domain)
1026
+ if _intake_hooks is not None:
1027
+ _intake_hooks.write_intake_artifact(ws.path, intake_record)
893
1028
  print(f"workspace created: {ws.path}")
894
1029
  print(f"manifest: {json.dumps(ws.load_manifest(), ensure_ascii=False, indent=2)}")
895
1030
  if args.dry_run:
@@ -897,9 +1032,9 @@ def _cmd_run(args: argparse.Namespace) -> int:
897
1032
  return 0
898
1033
  summary = advance(ws, demo_pack=args.demo_pack)
899
1034
  print(json.dumps(summary, ensure_ascii=False, indent=2))
900
- if summary["failures"]:
901
- return 1
902
- return 0
1035
+ if _intake_hooks is not None:
1036
+ _intake_hooks.after_run(ws.path, intake_record=intake_record, summary=summary)
1037
+ return 1 if summary["failures"] else 0
903
1038
 
904
1039
 
905
1040
  def _print_status(ws) -> None:
@@ -994,7 +1129,7 @@ def _cmd_domain(args) -> int:
994
1129
 
995
1130
  def _cmd_living(args) -> int:
996
1131
  from engine.project import ProjectWorkspace
997
- from engine.living import create_subscription, refresh, set_subscription_status
1132
+ from engine.living import create_subscription, refresh
998
1133
  home = _home(args)
999
1134
  ws = ProjectWorkspace.open(home, args.project)
1000
1135
  if args.action == "subscribe":
@@ -1014,7 +1149,6 @@ def _cmd_living(args) -> int:
1014
1149
  print(drift.get("summary", ""))
1015
1150
  return 0
1016
1151
  if args.action == "status":
1017
- from pathlib import Path as _P
1018
1152
  import json as _json
1019
1153
  p = ws.path / "living" / "subscriptions" / f"{args.subscription}.json"
1020
1154
  if not p.is_file():
@@ -1226,7 +1360,7 @@ def _cmd_study(args) -> int:
1226
1360
  print(e, file=sys.stderr)
1227
1361
  return 1
1228
1362
  from engine.study_design import save_study_design
1229
- path = save_study_design(ws, design)
1363
+ save_study_design(ws, design) # persists study-designs/<id>.json
1230
1364
  print(design["design_id"])
1231
1365
  return 0
1232
1366
 
@@ -1304,7 +1438,8 @@ def _cmd_migrate(args) -> int:
1304
1438
 
1305
1439
  def _cmd_dashboard(args) -> int:
1306
1440
  from dashboard_server import run_dashboard_server
1307
- run_dashboard_server(host=args.host, port=args.port)
1441
+ run_dashboard_server(host=args.host, port=args.port,
1442
+ open_browser=bool(getattr(args, "open", False)))
1308
1443
  return 0
1309
1444
 
1310
1445
 
@@ -1358,14 +1493,24 @@ def main(argv: list[str] | None = None) -> int:
1358
1493
  sub = parser.add_subparsers(dest="command", required=True)
1359
1494
 
1360
1495
  p_run = sub.add_parser("run", help="create a run workspace and advance stages")
1361
- p_run.add_argument("--question", required=True, help="education question to research")
1362
- p_run.add_argument("--depth", default="M", choices=["quick", "standard", "deep", "S", "M", "L"],
1363
- help="complexity depth (default: standard/M)")
1496
+ p_run.add_argument("--question", required=True, help="research question to investigate")
1497
+ p_run.add_argument("--domain", default="education", metavar="ID",
1498
+ help="registered research domain (default: education; "
1499
+ "see `domain list`)")
1500
+ p_run.add_argument("--depth", default=None,
1501
+ choices=["quick", "standard", "deep", "S", "M", "L", "auto"],
1502
+ help="complexity depth (S/M/L/auto or quick/standard/deep; "
1503
+ "default: intake/prefs auto guideline)")
1364
1504
  p_run.add_argument("--run-id", default=None, help="explicit run id (default: timestamp)")
1365
1505
  p_run.add_argument("--demo-pack", default=None, type=Path,
1366
1506
  help="seed external stages from an example pack (demo/test mode)")
1367
1507
  p_run.add_argument("--approve-agent-mcp", action="store_true",
1368
1508
  help="record agent-mcp approval in the manifest")
1509
+ p_run.add_argument("--yes", action="store_true",
1510
+ help="skip the two-round intake prompts; use prefs + flags unattended")
1511
+ p_run.add_argument("--enhancement", action="append", default=None,
1512
+ choices=["agent_mcp", "jev", "semdecide", "none"],
1513
+ help="execution enhancement (repeatable; used by intake)")
1369
1514
  p_run.add_argument("--dry-run", action="store_true", help="initialize only, do not advance")
1370
1515
  p_run.add_argument("--runs-dir", default=os.environ.get("EDUEVIDENCE_RUNS_DIR", str(ROOT / "runs")),
1371
1516
  help="directory holding run workspaces (default: <repo>/runs)")
@@ -1549,6 +1694,8 @@ def main(argv: list[str] | None = None) -> int:
1549
1694
  p_dash = sub.add_parser("dashboard", help="start local research & token dashboard")
1550
1695
  p_dash.add_argument("--host", default="127.0.0.1")
1551
1696
  p_dash.add_argument("--port", type=int, default=8765)
1697
+ p_dash.add_argument("--open", action="store_true",
1698
+ help="open the Studio URL in the system browser once the server is up")
1552
1699
  p_dash.set_defaults(func=_cmd_dashboard)
1553
1700
 
1554
1701
  p_srch = sub.add_parser("search", help="multi-channel hybrid search")