eduevidence 5.2.0 → 6.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (386) hide show
  1. package/CONTRIBUTING.md +105 -0
  2. package/README.md +142 -75
  3. package/README.zh-CN.md +73 -30
  4. package/SKILL.md +397 -131
  5. package/agents/openai.yaml +4 -0
  6. package/assets/readme/controlled-execution.svg +34 -0
  7. package/assets/readme/landing-tour.gif +0 -0
  8. package/assets/readme/logo.png +0 -0
  9. package/assets/readme/research-workflow.svg +56 -0
  10. package/assets/readme/studio-graph.png +0 -0
  11. package/assets/readme/studio-overview.png +0 -0
  12. package/assets/readme/studio-reports.png +0 -0
  13. package/assets/readme/studio-tour.gif +0 -0
  14. package/autoevolve/config.yaml +17 -0
  15. package/autoevolve/program.md +25 -0
  16. package/autoevolve/protected.manifest.yaml +34 -0
  17. package/benchmarks/adversarial/cases.jsonl +7 -0
  18. package/benchmarks/evidence-library.json +5268 -0
  19. package/benchmarks/partitions.json +8 -0
  20. package/bin/eduevidence.js +2 -1
  21. package/docs/architecture.md +496 -0
  22. package/docs/autoresearch-evolution-plan.md +2903 -0
  23. package/docs/autoresearch-implementation-status.md +101 -0
  24. package/docs/demo-storyboard.md +20 -0
  25. package/docs/demo-workplace-ai.md +92 -0
  26. package/docs/demo.md +32 -0
  27. package/docs/install-guide.md +150 -0
  28. package/docs/orchestration-role-model.md +1254 -0
  29. package/docs/release-closeout/README.md +17 -0
  30. package/docs/release-closeout/frontend-acceptance.md +23 -0
  31. package/docs/release-closeout/issues.md +19 -0
  32. package/docs/release-closeout/verification.md +28 -0
  33. package/docs/release-contract.md +108 -0
  34. package/docs/research-studio-guide.zh-CN.md +166 -0
  35. package/docs/sciverse-api.md +125 -0
  36. package/eduevidence_cli.py +29 -13
  37. package/engine/_resources.py +13 -0
  38. package/engine/autoevolve/__init__.py +3 -0
  39. package/engine/autoevolve/agent_view.py +167 -0
  40. package/engine/autoevolve/core.py +357 -0
  41. package/engine/autoevolve/events.py +11 -0
  42. package/engine/autoevolve/git_workspace.py +77 -0
  43. package/engine/autoevolve/projection.py +23 -0
  44. package/engine/autoevolve/runner.py +413 -0
  45. package/engine/autoevolve/trust.py +146 -0
  46. package/engine/autoresearch/__init__.py +6 -0
  47. package/engine/autoresearch/commit.py +132 -0
  48. package/engine/autoresearch/contracts.py +126 -0
  49. package/engine/autoresearch/controller.py +207 -0
  50. package/engine/autoresearch/events.py +12 -0
  51. package/engine/autoresearch/gap_priority.py +168 -0
  52. package/engine/autoresearch/projection.py +30 -0
  53. package/engine/autoresearch/research_memory.py +59 -0
  54. package/engine/autoresearch/saturation.py +91 -0
  55. package/engine/briefs.py +2 -1
  56. package/engine/capabilities.py +1 -0
  57. package/engine/contracts.py +3 -1
  58. package/engine/decision_policy.py +96 -0
  59. package/engine/evidence_graph.py +14 -10
  60. package/engine/evidencecore.py +7 -5
  61. package/engine/gaps.py +132 -73
  62. package/engine/ids.py +2 -0
  63. package/engine/judge_pack.py +65 -0
  64. package/engine/library.py +6 -2
  65. package/engine/library_builtin.py +3 -1
  66. package/engine/living.py +36 -5
  67. package/engine/meta_synthesis.py +3 -1
  68. package/engine/migration.py +88 -3
  69. package/engine/orchestration.py +460 -0
  70. package/engine/paths.py +2 -0
  71. package/engine/pilot.py +36 -33
  72. package/engine/project.py +2 -2
  73. package/engine/research_service.py +113 -0
  74. package/engine/studio_read_model.py +400 -0
  75. package/engine/taxonomy.py +211 -0
  76. package/engine/tribunal.py +44 -33
  77. package/engine/update.py +1 -0
  78. package/engine/versions.py +1 -1
  79. package/engine/worker_result.py +109 -0
  80. package/engine/workflows.py +70 -0
  81. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +2934 -0
  82. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
  83. package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
  84. package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
  85. package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
  86. package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
  87. package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
  88. package/examples/ai-coding-assistant-evidence/frame.json +48 -0
  89. package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
  90. package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
  91. package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
  92. package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
  93. package/examples/ai-coding-assistant-evidence/report_spec.json +230 -0
  94. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2934 -0
  95. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2934 -0
  96. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2934 -0
  97. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2934 -0
  98. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2934 -0
  99. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +2934 -0
  100. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +2934 -0
  101. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +2934 -0
  102. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +2934 -0
  103. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +2934 -0
  104. package/examples/ai-coding-assistant-evidence/result.json +1457 -0
  105. package/examples/ai-coding-assistant-evidence/result.zh.json +1457 -0
  106. package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
  107. package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
  108. package/examples/ai-coding-assistant-evidence/verdict.json +107 -0
  109. package/examples/spaced-retrieval-practice/applicability.json +14 -0
  110. package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
  111. package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
  112. package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
  113. package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
  114. package/examples/spaced-retrieval-practice/frame.json +58 -0
  115. package/examples/spaced-retrieval-practice/gate_report.json +101 -0
  116. package/examples/spaced-retrieval-practice/methodology.json +78 -0
  117. package/examples/spaced-retrieval-practice/report_spec.json +212 -0
  118. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
  119. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
  120. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
  121. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
  122. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
  123. package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
  124. package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
  125. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
  126. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
  127. package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
  128. package/examples/spaced-retrieval-practice/result.json +942 -0
  129. package/examples/spaced-retrieval-practice/result.zh.json +942 -0
  130. package/examples/spaced-retrieval-practice/skeptic.json +70 -0
  131. package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
  132. package/examples/spaced-retrieval-practice/verdict.json +93 -0
  133. package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
  134. package/examples/workplace-ai-assistant/claims.jsonl +4 -0
  135. package/examples/workplace-ai-assistant/evaluation.json +19 -0
  136. package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
  137. package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
  138. package/examples/workplace-ai-assistant/final_verdict.json +78 -0
  139. package/examples/workplace-ai-assistant/frame.json +41 -0
  140. package/examples/workplace-ai-assistant/gate_report.json +101 -0
  141. package/examples/workplace-ai-assistant/intervention.json +27 -0
  142. package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
  143. package/examples/workplace-ai-assistant/methodology.json +60 -0
  144. package/examples/workplace-ai-assistant/report_spec.json +224 -0
  145. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2814 -0
  146. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2814 -0
  147. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2814 -0
  148. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2814 -0
  149. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2814 -0
  150. package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
  151. package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
  152. package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
  153. package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
  154. package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
  155. package/examples/workplace-ai-assistant/result.json +615 -0
  156. package/examples/workplace-ai-assistant/result.zh.json +615 -0
  157. package/examples/workplace-ai-assistant/search_log.json +19 -0
  158. package/examples/workplace-ai-assistant/skeptic.json +72 -0
  159. package/examples/workplace-ai-assistant/sources.jsonl +3 -0
  160. package/examples/workplace-ai-assistant/validation_result.json +9 -0
  161. package/examples/workplace-ai-assistant/verdict.json +78 -0
  162. package/install.sh +7 -7
  163. package/integrations/agent_mcp.py +2 -2
  164. package/integrations/orchestration_dispatch.py +146 -0
  165. package/package.json +46 -3
  166. package/pyproject.toml +14 -22
  167. package/references/autoresearch.md +30 -0
  168. package/references/evaluation-policy.md +24 -0
  169. package/references/orchestration.md +22 -0
  170. package/references/report-copy-style.md +67 -0
  171. package/references/retrieval-compliance.md +75 -0
  172. package/references/retrieval-protocol.md +20 -0
  173. package/references/scientific-invariants.md +19 -0
  174. package/retrieval/audit.py +178 -0
  175. package/retrieval/fetch.py +96 -0
  176. package/retrieval/sciverse.py +398 -0
  177. package/retrieval/search.py +47 -7
  178. package/schemas/applicability.schema.json +94 -0
  179. package/schemas/chart-spec.schema.json +10 -3
  180. package/schemas/evidence.schema.json +316 -43
  181. package/schemas/fetch-result.schema.json +2 -1
  182. package/schemas/intervention.schema.json +106 -21
  183. package/schemas/report-result.schema.json +12 -4
  184. package/schemas/report-spec.schema.json +98 -100
  185. package/schemas/skeptic.schema.json +86 -0
  186. package/schemas/source.schema.json +21 -2
  187. package/schemas/v2/finding.schema.json +5 -1
  188. package/schemas/v2/methodology-audit.schema.json +5 -1
  189. package/schemas/v2/outcome.schema.json +28 -5
  190. package/schemas/v2/project.schema.json +2 -2
  191. package/schemas/v2/run.schema.json +1 -1
  192. package/schemas/v2/study.schema.json +5 -1
  193. package/schemas/vNext/autoevolve-session.schema.json +34 -0
  194. package/schemas/vNext/eval-snapshot.schema.json +77 -0
  195. package/schemas/vNext/execution-plan.schema.json +50 -0
  196. package/schemas/vNext/gap-priority.schema.json +54 -0
  197. package/schemas/vNext/negative-search-record.schema.json +68 -0
  198. package/schemas/vNext/research-iteration.schema.json +87 -0
  199. package/schemas/vNext/research-strategy.schema.json +62 -0
  200. package/schemas/vNext/skill-experiment.schema.json +90 -0
  201. package/schemas/vNext/task-spec.schema.json +156 -0
  202. package/schemas/vNext/worker-result.schema.json +60 -0
  203. package/schemas/verdict.schema.json +164 -28
  204. package/scripts/benchmark_judge.py +2 -2
  205. package/scripts/benchmark_v3.py +26 -43
  206. package/scripts/build_esl_artifacts.py +4 -4
  207. package/scripts/build_evidence_library.py +2 -2
  208. package/scripts/build_gh_pages.py +98 -0
  209. package/scripts/build_readme_diagrams.py +72 -0
  210. package/scripts/build_report_variants.py +101 -0
  211. package/scripts/build_result.py +74 -9
  212. package/scripts/check_autoresearch_invariants.py +95 -0
  213. package/scripts/check_package_parity.py +85 -0
  214. package/scripts/check_protocol_alignment.py +375 -0
  215. package/scripts/check_versioned_schemas.py +254 -0
  216. package/scripts/claim_audit.py +13 -8
  217. package/scripts/compute_confidence.py +10 -0
  218. package/scripts/daily_evolve.py +30 -0
  219. package/scripts/dashboard_server.py +130 -101
  220. package/scripts/did_regression.py +17 -32
  221. package/scripts/enrich_projects_human_and_lieflat.py +1 -1
  222. package/scripts/evidence_score.py +5 -2
  223. package/scripts/generate_metrics.py +4 -3
  224. package/scripts/generate_new_projects.py +5 -5
  225. package/scripts/orchestrator.py +286 -36
  226. package/scripts/pre_verdict_gate.py +224 -26
  227. package/scripts/quickstart.py +18 -2
  228. package/scripts/rebake_all_5themes.py +1 -2
  229. package/scripts/research_auto_cli.py +475 -0
  230. package/scripts/run_workspace.py +24 -8
  231. package/scripts/search_provenance.py +64 -0
  232. package/scripts/serve_web.py +9 -10
  233. package/scripts/skill_lint.py +1 -1
  234. package/scripts/skill_payload.py +81 -0
  235. package/scripts/test_adversarial_empirical.py +26 -19
  236. package/scripts/validate_schema.py +46 -2
  237. package/scripts/vnext_cli.py +133 -0
  238. package/setup.py +12 -0
  239. package/skill/agents/evaluation-designer.md +20 -4
  240. package/skill/agents/evidence-analyst.md +19 -3
  241. package/skill/agents/evidence-judge.md +50 -2
  242. package/skill/agents/evidence-retriever.md +20 -3
  243. package/skill/agents/intervention-designer.md +20 -4
  244. package/skill/agents/method-reviewer.md +18 -2
  245. package/skill/agents/{education-planner.md → research-planner.md} +19 -3
  246. package/skill/agents/skeptic.md +18 -2
  247. package/skill/roles/registry.yaml +45 -0
  248. package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
  249. package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
  250. package/skill/sub-skills/data-analysis/SKILL.md +34 -15
  251. package/skill/sub-skills/ethics-review/SKILL.md +33 -10
  252. package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
  253. package/skill/sub-skills/evidence-review/SKILL.md +31 -12
  254. package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
  255. package/skill/sub-skills/literature-review/SKILL.md +35 -14
  256. package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
  257. package/skill/sub-skills/report-generation/SKILL.md +40 -6
  258. package/skill/sub-skills/research-planning/SKILL.md +41 -14
  259. package/skill/sub-skills/study-design/SKILL.md +30 -9
  260. package/skill/task-briefs/adjudicate.md +32 -7
  261. package/skill/task-briefs/applicability.md +38 -0
  262. package/skill/task-briefs/audit.md +32 -7
  263. package/skill/task-briefs/challenge.md +34 -5
  264. package/skill/task-briefs/evaluate.md +30 -5
  265. package/skill/task-briefs/extract.md +31 -8
  266. package/skill/task-briefs/frame.md +39 -10
  267. package/skill/task-briefs/intervene.md +32 -6
  268. package/skill/task-briefs/present.md +32 -8
  269. package/skill/task-briefs/projection.md +37 -0
  270. package/skill/task-briefs/retrieve.md +36 -6
  271. package/skill/workflows/decision-and-pilot.md +85 -0
  272. package/skill/workflows/evaluate-and-update.md +93 -0
  273. package/skill/workflows/evidence-review.md +117 -0
  274. package/visualization/eduevidence-report/assets/base.css +2 -2
  275. package/visualization/eduevidence-report/assets/reader.css +752 -0
  276. package/visualization/eduevidence-report/assets/reader.js +132 -0
  277. package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
  278. package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
  279. package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
  280. package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
  281. package/visualization/eduevidence-report/scripts/build_report.py +561 -121
  282. package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
  283. package/visualization/eduevidence-report/scripts/lieflat_engine.py +371 -136
  284. package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
  285. package/visualization/eduevidence-report/themes/academic.css +1 -1
  286. package/visualization/eduevidence-report/themes/claude.css +1 -1
  287. package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
  288. package/visualization/eduevidence-report/themes/datalab.css +2 -2
  289. package/visualization/eduevidence-report/themes/presentation.css +2 -2
  290. package/web/README.md +18 -0
  291. package/web/architecture.html +14885 -0
  292. package/web/index.html +53 -0
  293. package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
  294. package/web/studio/assets/index-B8tkF44Q.css +1 -0
  295. package/web/studio/assets/index-CQ6Keoyc.js +230 -0
  296. package/web/studio/config.json +1 -0
  297. package/web/studio/index.html +14 -0
  298. package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
  299. package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
  300. package/engine/__pycache__/bias.cpython-312.pyc +0 -0
  301. package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
  302. package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
  303. package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
  304. package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
  305. package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
  306. package/engine/__pycache__/events.cpython-312.pyc +0 -0
  307. package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
  308. package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
  309. package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
  310. package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
  311. package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
  312. package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
  313. package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
  314. package/engine/__pycache__/ids.cpython-312.pyc +0 -0
  315. package/engine/__pycache__/library.cpython-312.pyc +0 -0
  316. package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
  317. package/engine/__pycache__/living.cpython-312.pyc +0 -0
  318. package/engine/__pycache__/log.cpython-312.pyc +0 -0
  319. package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
  320. package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
  321. package/engine/__pycache__/migration.cpython-312.pyc +0 -0
  322. package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
  323. package/engine/__pycache__/paths.cpython-312.pyc +0 -0
  324. package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
  325. package/engine/__pycache__/planner.cpython-312.pyc +0 -0
  326. package/engine/__pycache__/project.cpython-312.pyc +0 -0
  327. package/engine/__pycache__/projections.cpython-312.pyc +0 -0
  328. package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
  329. package/engine/__pycache__/run.cpython-312.pyc +0 -0
  330. package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
  331. package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
  332. package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
  333. package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
  334. package/engine/__pycache__/update.cpython-312.pyc +0 -0
  335. package/engine/__pycache__/versions.cpython-312.pyc +0 -0
  336. package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
  337. package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
  338. package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
  339. package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
  340. package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
  341. package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
  342. package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
  343. package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
  344. package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
  345. package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
  346. package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
  347. package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
  348. package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
  349. package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
  350. package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
  351. package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
  352. package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
  353. package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
  354. package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
  355. package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
  356. package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
  357. package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
  358. package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
  359. package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
  360. package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
  361. package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
  362. package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
  363. package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
  364. package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
  365. package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
  366. package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
  367. package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
  368. package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
  369. package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
  370. package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
  371. package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
  372. package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
  373. package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
  374. package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
  375. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
  376. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
  377. package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
  378. package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
  379. package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
  380. package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
  381. package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
  382. package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
  383. package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
  384. package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
  385. package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
  386. package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
@@ -0,0 +1,460 @@
1
+ """Canonical orchestration primitives for EduEvidence.
2
+
3
+ Protocol stage, scientific role, capability, worker/subagent, and model/CLI are
4
+ separate concepts. A RoleSpec describes scientific responsibility; a TaskSpec
5
+ is one bounded execution contract; an ExecutionPlan states sequencing and
6
+ parallel groups. Runtime model/CLI selection remains an adapter concern.
7
+
8
+ Scientific rule: workers produce staging artifacts only. Canonical project
9
+ state is committed by the single-writer lead path after validation.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import json
14
+ from dataclasses import asdict, dataclass, field
15
+ from enum import Enum
16
+ from typing import Any, Iterable
17
+
18
+
19
+ class Complexity(str, Enum):
20
+ S = "S"
21
+ M = "M"
22
+ L = "L"
23
+
24
+
25
+ class ExecutionMode(str, Enum):
26
+ LOCAL = "local"
27
+ DELEGATED = "delegated"
28
+
29
+
30
+ CANONICAL_STAGES = (
31
+ "frame",
32
+ "retrieve",
33
+ "extract",
34
+ "challenge",
35
+ "audit",
36
+ "adjudicate",
37
+ "applicability",
38
+ "intervene",
39
+ "evaluate",
40
+ )
41
+
42
+ CANONICAL_STATE_ARTIFACTS = frozenset({
43
+ "EvidenceGraph",
44
+ "GraphRevision",
45
+ "DecisionSnapshot",
46
+ "KnowledgeGap",
47
+ "StudyDesign",
48
+ "PilotRun",
49
+ "AnalysisRun",
50
+ })
51
+
52
+ DEFAULT_FORBIDDEN_WORKER_ACTIONS = (
53
+ "canonical_state_write",
54
+ "decision_promotion",
55
+ "recursive_worker_spawn",
56
+ "evaluator_mutation",
57
+ )
58
+
59
+
60
+ @dataclass(frozen=True)
61
+ class RoleSpec:
62
+ """A scientific responsibility, not an instruction to spawn an agent."""
63
+
64
+ name: str
65
+ responsibility: str
66
+ stages: tuple[str, ...]
67
+ capabilities: tuple[str, ...]
68
+ independence_required: bool = False
69
+ critical_path: bool = False
70
+
71
+ def __post_init__(self) -> None:
72
+ unknown = set(self.stages) - set(CANONICAL_STAGES)
73
+ if unknown:
74
+ raise ValueError(f"role {self.name!r} references unknown stages: {sorted(unknown)}")
75
+
76
+
77
+ ROLE_REGISTRY: dict[str, RoleSpec] = {
78
+ "research-planner": RoleSpec(
79
+ "research-planner",
80
+ "Own framing completeness, scope, comparison and outcome definition.",
81
+ ("frame",),
82
+ ("research-planning",),
83
+ critical_path=True,
84
+ ),
85
+ "evidence-retriever": RoleSpec(
86
+ "evidence-retriever",
87
+ "Acquire candidate sources and record verifiable provenance without adjudicating them.",
88
+ ("retrieve",),
89
+ ("literature-review", "full-text-fetch", "source-validation"),
90
+ ),
91
+ "evidence-analyst": RoleSpec(
92
+ "evidence-analyst",
93
+ "Extract structured study findings without deciding what the evidence body means.",
94
+ ("extract",),
95
+ ("evidence-extraction",),
96
+ ),
97
+ "skeptic": RoleSpec(
98
+ "skeptic",
99
+ "Own independent counter-evidence and alternative-explanation coverage.",
100
+ ("challenge",),
101
+ ("contradiction-analysis", "counter-retrieval"),
102
+ independence_required=True,
103
+ critical_path=True,
104
+ ),
105
+ "method-reviewer": RoleSpec(
106
+ "method-reviewer",
107
+ "Own study-level methodology appraisal and evidence-quality limitations.",
108
+ ("audit",),
109
+ ("methodology-audit",),
110
+ independence_required=True,
111
+ critical_path=True,
112
+ ),
113
+ "evidence-judge": RoleSpec(
114
+ "evidence-judge",
115
+ "Own the bounded decision after evidence, challenge and audit gates pass.",
116
+ ("adjudicate", "applicability"),
117
+ ("evidence-review", "decision-adjudication", "applicability"),
118
+ critical_path=True,
119
+ ),
120
+ "intervention-designer": RoleSpec(
121
+ "intervention-designer",
122
+ "Turn a grounded decision and KnowledgeGap into a bounded intervention or pilot.",
123
+ ("intervene",),
124
+ ("study-design", "intervention-design"),
125
+ ),
126
+ "evaluation-designer": RoleSpec(
127
+ "evaluation-designer",
128
+ "Define estimable evaluation, retention/transfer measurement and update logic.",
129
+ ("evaluate",),
130
+ ("evaluation-design", "data-analysis"),
131
+ ),
132
+ }
133
+
134
+
135
+ @dataclass(frozen=True)
136
+ class TaskSpec:
137
+ """One bounded scientific task; delegation requires full runtime context."""
138
+
139
+ task_id: str
140
+ stage: str
141
+ role: str
142
+ objective: str
143
+ evidence_axis: str
144
+ run_id: str | None = None
145
+ base_revision: int | None = None
146
+ role_profile: str | None = None
147
+ reason_for_delegation: str | None = None
148
+ inputs: tuple[str, ...] = () # legacy alias kept during migration
149
+ input_artifacts: tuple[str, ...] = ()
150
+ allowed_capabilities: tuple[str, ...] = ()
151
+ forbidden_actions: tuple[str, ...] = DEFAULT_FORBIDDEN_WORKER_ACTIONS
152
+ scope: dict[str, Any] = field(default_factory=dict)
153
+ budget: dict[str, Any] = field(default_factory=dict)
154
+ expected_outputs: tuple[str, ...] = ()
155
+ output_contract: dict[str, Any] = field(default_factory=dict)
156
+ termination: dict[str, Any] = field(default_factory=dict)
157
+ execution_mode: ExecutionMode = ExecutionMode.LOCAL
158
+ independent: bool = False
159
+ read_only: bool = True
160
+ timeout_seconds: int = 1800
161
+ token_budget: int | None = None
162
+ metadata: dict[str, Any] = field(default_factory=dict)
163
+
164
+ def validate(self) -> None:
165
+ if not self.task_id.strip():
166
+ raise ValueError("task_id must be non-empty")
167
+ if self.stage not in CANONICAL_STAGES:
168
+ raise ValueError(f"unknown protocol stage: {self.stage!r}")
169
+ spec = ROLE_REGISTRY.get(self.role)
170
+ if spec is None:
171
+ raise ValueError(f"unknown scientific role: {self.role!r}")
172
+ if self.stage not in spec.stages:
173
+ raise ValueError(
174
+ f"role {self.role!r} does not own stage {self.stage!r}; owns {spec.stages}"
175
+ )
176
+ if self.role_profile not in (None, self.role):
177
+ raise ValueError("role_profile must identify the same scientific role")
178
+ if not self.objective.strip():
179
+ raise ValueError("objective must be non-empty")
180
+ if not self.evidence_axis.strip():
181
+ raise ValueError("evidence_axis must be non-empty")
182
+ if self.base_revision is not None and self.base_revision < 0:
183
+ raise ValueError("base_revision must be >= 0")
184
+ if self.timeout_seconds <= 0:
185
+ raise ValueError("timeout_seconds must be positive")
186
+ if self.token_budget is not None and self.token_budget <= 0:
187
+ raise ValueError("token_budget must be positive when provided")
188
+ unknown_caps = set(self.allowed_capabilities) - set(spec.capabilities)
189
+ if unknown_caps:
190
+ raise ValueError(
191
+ f"task grants capabilities outside role {self.role!r}: {sorted(unknown_caps)}"
192
+ )
193
+ forbidden = CANONICAL_STATE_ARTIFACTS.intersection(self.expected_outputs)
194
+ if self.execution_mode is ExecutionMode.DELEGATED:
195
+ if not self.read_only:
196
+ raise ValueError("delegated workers must be read-only against canonical state")
197
+ if forbidden:
198
+ raise ValueError(
199
+ "delegated workers may only return staging artifacts; canonical outputs forbidden: "
200
+ f"{sorted(forbidden)}"
201
+ )
202
+ if "canonical_state_write" not in self.forbidden_actions:
203
+ raise ValueError("delegated TaskSpec must explicitly forbid canonical_state_write")
204
+ if spec.independence_required and not self.independent:
205
+ raise ValueError(f"role {self.role!r} requires independent delegated execution")
206
+
207
+ def validate_for_dispatch(self) -> None:
208
+ self.validate()
209
+ if self.execution_mode is not ExecutionMode.DELEGATED:
210
+ raise ValueError("only delegated TaskSpecs may be dispatched")
211
+ if not self.run_id or not self.run_id.strip():
212
+ raise ValueError("delegated TaskSpec requires run_id")
213
+ if self.base_revision is None:
214
+ raise ValueError("delegated TaskSpec requires base_revision")
215
+ if not self.reason_for_delegation or not self.reason_for_delegation.strip():
216
+ raise ValueError("delegated TaskSpec requires reason_for_delegation")
217
+ if not self.allowed_capabilities:
218
+ raise ValueError("delegated TaskSpec requires explicit allowed_capabilities")
219
+ if not self.output_contract:
220
+ raise ValueError("delegated TaskSpec requires output_contract")
221
+ if not self.termination:
222
+ raise ValueError("delegated TaskSpec requires termination contract")
223
+
224
+ def to_dict(self) -> dict[str, Any]:
225
+ value = asdict(self)
226
+ value["execution_mode"] = self.execution_mode.value
227
+ return value
228
+
229
+ def to_prompt_contract(self) -> str:
230
+ """Produce a stable, explicit worker envelope."""
231
+ self.validate_for_dispatch()
232
+ return (
233
+ f"TASK_ID: {self.task_id}\n"
234
+ f"RUN_ID: {self.run_id}\n"
235
+ f"BASE_GRAPH_REVISION: {self.base_revision}\n"
236
+ f"STAGE: {self.stage}\n"
237
+ f"SCIENTIFIC_ROLE: {self.role}\n"
238
+ f"EVIDENCE_AXIS: {self.evidence_axis}\n"
239
+ f"OBJECTIVE: {self.objective}\n"
240
+ f"REASON_FOR_DELEGATION: {self.reason_for_delegation}\n"
241
+ f"ALLOWED_CAPABILITIES: {', '.join(self.allowed_capabilities)}\n"
242
+ f"FORBIDDEN_ACTIONS: {', '.join(self.forbidden_actions)}\n"
243
+ f"INPUT_ARTIFACTS: {', '.join(self.input_artifacts or self.inputs) if (self.input_artifacts or self.inputs) else 'none'}\n"
244
+ f"SCOPE_JSON: {json.dumps(self.scope, ensure_ascii=False, sort_keys=True)}\n"
245
+ f"BUDGET_JSON: {json.dumps(self.budget, ensure_ascii=False, sort_keys=True)}\n"
246
+ f"OUTPUT_CONTRACT_JSON: {json.dumps(self.output_contract, ensure_ascii=False, sort_keys=True)}\n"
247
+ f"TERMINATION_JSON: {json.dumps(self.termination, ensure_ascii=False, sort_keys=True)}\n"
248
+ "CANONICAL_STATE_WRITE: FORBIDDEN\n"
249
+ "Return only the requested staging artifact(s)."
250
+ )
251
+
252
+
253
+ @dataclass(frozen=True)
254
+ class ExecutionPlan:
255
+ complexity: Complexity
256
+ tasks: tuple[TaskSpec, ...]
257
+ max_parallel_workers: int
258
+ parallel_groups: tuple[tuple[str, ...], ...] = ()
259
+ plan_id: str | None = None
260
+
261
+ @property
262
+ def delegated_tasks(self) -> tuple[TaskSpec, ...]:
263
+ return tuple(t for t in self.tasks if t.execution_mode is ExecutionMode.DELEGATED)
264
+
265
+ def validate(self) -> None:
266
+ if self.max_parallel_workers < 0:
267
+ raise ValueError("max_parallel_workers cannot be negative")
268
+ if self.max_parallel_workers > 6:
269
+ raise ValueError("bounded worker pool exceeds hard limit of 6")
270
+ ids: set[str] = set()
271
+ by_id: dict[str, TaskSpec] = {}
272
+ for task in self.tasks:
273
+ task.validate()
274
+ if task.task_id in ids:
275
+ raise ValueError(f"duplicate task id: {task.task_id}")
276
+ ids.add(task.task_id)
277
+ by_id[task.task_id] = task
278
+ if self.delegated_tasks and self.max_parallel_workers < 1:
279
+ raise ValueError("delegated plan requires max_parallel_workers >= 1")
280
+
281
+ grouped: list[str] = []
282
+ for group in self.parallel_groups:
283
+ if not group:
284
+ raise ValueError("parallel groups may not be empty")
285
+ if len(group) > self.max_parallel_workers:
286
+ raise ValueError("parallel group exceeds max_parallel_workers")
287
+ for task_id in group:
288
+ task = by_id.get(task_id)
289
+ if task is None:
290
+ raise ValueError(f"parallel group references unknown task {task_id}")
291
+ if task.execution_mode is not ExecutionMode.DELEGATED:
292
+ raise ValueError(f"parallel group may contain delegated tasks only: {task_id}")
293
+ grouped.append(task_id)
294
+ delegated_ids = [task.task_id for task in self.delegated_tasks]
295
+ if sorted(grouped) != sorted(delegated_ids):
296
+ raise ValueError("every delegated task must appear exactly once in parallel_groups")
297
+
298
+ def to_dict(self) -> dict[str, Any]:
299
+ return {
300
+ "plan_id": self.plan_id,
301
+ "complexity": self.complexity.value,
302
+ "tasks": [task.to_dict() for task in self.tasks],
303
+ "max_parallel_workers": self.max_parallel_workers,
304
+ "parallel_groups": [list(group) for group in self.parallel_groups],
305
+ }
306
+
307
+
308
+ class ExecutionPlanner:
309
+ """Deterministic policy planner; it never chooses a concrete model/CLI."""
310
+
311
+ def plan(
312
+ self,
313
+ complexity: str | Complexity,
314
+ *,
315
+ run_id: str | None = None,
316
+ base_revision: int | None = None,
317
+ plan_id: str | None = None,
318
+ ) -> ExecutionPlan:
319
+ level = complexity if isinstance(complexity, Complexity) else Complexity(complexity.upper())
320
+ if level is Complexity.S:
321
+ plan = ExecutionPlan(
322
+ level,
323
+ self._serial_tasks(run_id, base_revision),
324
+ max_parallel_workers=0,
325
+ parallel_groups=(),
326
+ plan_id=plan_id,
327
+ )
328
+ elif level is Complexity.M:
329
+ tasks = self._medium_tasks(run_id, base_revision)
330
+ plan = ExecutionPlan(
331
+ level,
332
+ tasks,
333
+ max_parallel_workers=2,
334
+ parallel_groups=(("retrieve-direct", "retrieve-counter"), ("challenge",)),
335
+ plan_id=plan_id,
336
+ )
337
+ else:
338
+ tasks = self._deep_tasks(run_id, base_revision)
339
+ plan = ExecutionPlan(
340
+ level,
341
+ tasks,
342
+ max_parallel_workers=4,
343
+ parallel_groups=(
344
+ (
345
+ "retrieve-direct",
346
+ "retrieve-transfer",
347
+ "retrieve-counter",
348
+ "retrieve-applicability",
349
+ ),
350
+ ("challenge", "audit"),
351
+ ),
352
+ plan_id=plan_id,
353
+ )
354
+ plan.validate()
355
+ return plan
356
+
357
+ @staticmethod
358
+ def _base(
359
+ task_id: str,
360
+ stage: str,
361
+ role: str,
362
+ objective: str,
363
+ axis: str,
364
+ *,
365
+ run_id: str | None,
366
+ base_revision: int | None,
367
+ delegated: bool = False,
368
+ independent: bool = False,
369
+ outputs: Iterable[str] = (),
370
+ ) -> TaskSpec:
371
+ role_spec = ROLE_REGISTRY[role]
372
+ timeout_seconds = 1800
373
+ token_budget = None
374
+ outputs = tuple(outputs)
375
+ return TaskSpec(
376
+ task_id=task_id,
377
+ stage=stage,
378
+ role=role,
379
+ role_profile=role,
380
+ objective=objective,
381
+ evidence_axis=axis,
382
+ run_id=run_id,
383
+ base_revision=base_revision,
384
+ reason_for_delegation=(
385
+ f"independent bounded {axis} work benefits from delegated context"
386
+ if delegated
387
+ else None
388
+ ),
389
+ allowed_capabilities=role_spec.capabilities,
390
+ forbidden_actions=DEFAULT_FORBIDDEN_WORKER_ACTIONS,
391
+ scope={"evidence_axis": axis},
392
+ budget={"timeout_seconds": timeout_seconds, "token_budget": token_budget},
393
+ expected_outputs=outputs,
394
+ output_contract={
395
+ "artifact_types": list(outputs),
396
+ "canonical_state": False,
397
+ "validation_owner": "lead-orchestrator",
398
+ },
399
+ termination={
400
+ "max_seconds": timeout_seconds,
401
+ "conditions": ["output_contract_satisfied", "budget_exhausted", "tool_failure"],
402
+ },
403
+ execution_mode=ExecutionMode.DELEGATED if delegated else ExecutionMode.LOCAL,
404
+ independent=independent,
405
+ read_only=True,
406
+ timeout_seconds=timeout_seconds,
407
+ token_budget=token_budget,
408
+ )
409
+
410
+ def _serial_tasks(self, run_id, base_revision) -> tuple[TaskSpec, ...]:
411
+ return (
412
+ self._base("frame", "frame", "research-planner", "Structure the research question.", "frame", run_id=run_id, base_revision=base_revision),
413
+ self._base("retrieve", "retrieve", "evidence-retriever", "Acquire bounded evidence.", "direct+counter", run_id=run_id, base_revision=base_revision),
414
+ self._base("extract", "extract", "evidence-analyst", "Extract structured findings.", "all-eligible", run_id=run_id, base_revision=base_revision),
415
+ self._base("challenge", "challenge", "skeptic", "Challenge the provisional interpretation.", "counter-evidence", run_id=run_id, base_revision=base_revision),
416
+ self._base("audit", "audit", "method-reviewer", "Audit methodology and outcome validity.", "methodology", run_id=run_id, base_revision=base_revision),
417
+ self._base("judge", "adjudicate", "evidence-judge", "Emit an evidence-bounded decision.", "decision", run_id=run_id, base_revision=base_revision),
418
+ )
419
+
420
+ def _medium_tasks(self, run_id, base_revision) -> tuple[TaskSpec, ...]:
421
+ return (
422
+ self._base("frame", "frame", "research-planner", "Structure the research question.", "frame", run_id=run_id, base_revision=base_revision),
423
+ self._base("retrieve-direct", "retrieve", "evidence-retriever", "Retrieve direct decision-relevant evidence.", "direct-causal", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
424
+ self._base("retrieve-counter", "retrieve", "evidence-retriever", "Retrieve null, negative and contradictory evidence.", "counter-risk", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
425
+ self._base("extract", "extract", "evidence-analyst", "Merge validated sources and extract findings.", "all-eligible", run_id=run_id, base_revision=base_revision),
426
+ self._base("challenge", "challenge", "skeptic", "Independently test the provisional interpretation.", "counter-evidence", run_id=run_id, base_revision=base_revision, delegated=True, independent=True, outputs=("SkepticFindings",)),
427
+ self._base("audit", "audit", "method-reviewer", "Audit methodology and construct validity.", "methodology", run_id=run_id, base_revision=base_revision),
428
+ self._base("judge", "adjudicate", "evidence-judge", "Emit an evidence-bounded decision.", "decision", run_id=run_id, base_revision=base_revision),
429
+ )
430
+
431
+ def _deep_tasks(self, run_id, base_revision) -> tuple[TaskSpec, ...]:
432
+ return (
433
+ self._base("frame", "frame", "research-planner", "Structure the research question.", "frame", run_id=run_id, base_revision=base_revision),
434
+ self._base("retrieve-direct", "retrieve", "evidence-retriever", "Retrieve direct causal evidence.", "direct-causal", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
435
+ self._base("retrieve-transfer", "retrieve", "evidence-retriever", "Retrieve retention and independent-transfer evidence.", "transfer-retention", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
436
+ self._base("retrieve-counter", "retrieve", "evidence-retriever", "Retrieve null, negative, risk and contradiction evidence.", "counter-risk", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
437
+ self._base("retrieve-applicability", "retrieve", "evidence-retriever", "Retrieve subgroup, context and freshness evidence.", "applicability-freshness", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
438
+ self._base("extract", "extract", "evidence-analyst", "Deterministically merge and extract validated findings.", "all-eligible", run_id=run_id, base_revision=base_revision),
439
+ self._base("challenge", "challenge", "skeptic", "Independently challenge the merged interpretation.", "counter-evidence", run_id=run_id, base_revision=base_revision, delegated=True, independent=True, outputs=("SkepticFindings",)),
440
+ self._base("audit", "audit", "method-reviewer", "Independently audit methodology and outcome validity.", "methodology", run_id=run_id, base_revision=base_revision, delegated=True, independent=True, outputs=("MethodologyAudit",)),
441
+ self._base("judge", "adjudicate", "evidence-judge", "Emit an evidence-bounded decision after all gates.", "decision", run_id=run_id, base_revision=base_revision),
442
+ )
443
+
444
+
445
+ class CanonicalWriteGuard:
446
+ """Fail-closed single-writer authority for canonical state mutations."""
447
+
448
+ def __init__(self, writer_id: str = "lead-orchestrator") -> None:
449
+ if not writer_id.strip():
450
+ raise ValueError("writer_id must be non-empty")
451
+ self.writer_id = writer_id
452
+
453
+ def require(self, actor_id: str, artifact_type: str) -> None:
454
+ if artifact_type not in CANONICAL_STATE_ARTIFACTS:
455
+ return
456
+ if actor_id != self.writer_id:
457
+ raise PermissionError(
458
+ f"canonical state {artifact_type} may only be written by {self.writer_id!r}; "
459
+ f"actor {actor_id!r} is staging-only"
460
+ )
package/engine/paths.py CHANGED
@@ -4,6 +4,8 @@ EDUEVIDENCE_HOME (default `~/.eduevidence`) is the root that owns the Shared
4
4
  Research Library and all Projects. An explicit path always wins.
5
5
  """
6
6
 
7
+ from __future__ import annotations
8
+
7
9
  from pathlib import Path
8
10
  import os
9
11
 
package/engine/pilot.py CHANGED
@@ -22,6 +22,7 @@ from engine.datasets import analysis_blocked_by_privacy, derive_csv_profile, ing
22
22
  from engine.graph_store import GraphMutation, GraphStore
23
23
  from engine.ids import new_local_id, new_run_id
24
24
  from engine.project import ProjectWorkspace
25
+ from engine.taxonomy import category_of, tokens as taxonomy_tokens
25
26
  from engine.synthesis import synthesize_project
26
27
  from engine.tribunal import adjudicate, decision_diff, save_decision_snapshot
27
28
  from engine.versions import (
@@ -30,15 +31,21 @@ from engine.versions import (
30
31
  SOURCE_VALIDATION_POLICY_VERSION,
31
32
  )
32
33
 
33
- #: Outcome Taxonomy tokens that pilots may measure (mirrors outcome-taxonomy.md).
34
- OUTCOME_TAXONOMY = {
35
- "knowledge_gain", "concept_understanding", "retention", "transfer",
36
- "independent_problem_solving", "completion_time", "accuracy",
37
- "code_quality", "assignment_score", "engagement", "motivation",
38
- "cognitive_load", "help_seeking", "metacognition", "ai_dependency",
39
- "over_reliance", "reduced_effort", "reduced_transfer",
40
- "academic_integrity_risk", "false_confidence",
41
- }
34
+ def outcome_taxonomy_tokens(domain: str = "education") -> set[str]:
35
+ """Outcome tokens a pilot may measure, read from the domain registry.
36
+
37
+ This was a module-level set of the 20 education tokens, so a policy pilot
38
+ could not register at all. The domain registry (via engine/taxonomy.py) is
39
+ now the single authority.
40
+ """
41
+ return set(taxonomy_tokens(domain))
42
+
43
+
44
+ def project_domain(project: ProjectWorkspace) -> str:
45
+ """The domain a project registers; defaults to education when absent."""
46
+ manifest = project.manifest()
47
+ return str(manifest.get("domain") or "education")
48
+
42
49
 
43
50
  PILOT_STATUSES = ("registered", "data_imported", "analyzed", "adjudicated")
44
51
 
@@ -49,22 +56,6 @@ PII_COLUMN_HINTS = ("name", "student", "学号", "姓名", "email", "mail",
49
56
  _DECISION_IMPLICATION = {"support": "support_adoption",
50
57
  "contradict": "oppose_adoption", "neutral": "neutral"}
51
58
 
52
- #: Outcome Taxonomy token -> graph outcome category enum (schemas/v2/outcome).
53
- _OUTCOME_CATEGORY = {
54
- "knowledge_gain": "learning", "concept_understanding": "learning",
55
- "retention": "learning", "transfer": "learning",
56
- "independent_problem_solving": "learning",
57
- "completion_time": "task_performance", "accuracy": "task_performance",
58
- "code_quality": "task_performance", "assignment_score": "task_performance",
59
- "engagement": "process", "motivation": "process",
60
- "cognitive_load": "process", "help_seeking": "process",
61
- "metacognition": "process",
62
- "ai_dependency": "risk", "over_reliance": "risk",
63
- "reduced_effort": "risk", "reduced_transfer": "risk",
64
- "academic_integrity_risk": "risk", "false_confidence": "risk",
65
- }
66
-
67
-
68
59
  def _now_iso() -> str:
69
60
  return datetime.now(timezone.utc).isoformat()
70
61
 
@@ -83,7 +74,8 @@ def _load_pilot(project: ProjectWorkspace, pilot_id: str) -> dict:
83
74
  def _save_pilot(project: ProjectWorkspace, pilot: dict) -> Path:
84
75
  from scripts.validate_schema import SchemaError, validate # noqa: PLC0415
85
76
 
86
- schema_path = (Path(__file__).resolve().parent.parent / "schemas" / "v3"
77
+ from engine._resources import resource_root
78
+ schema_path = (resource_root() / "schemas" / "v3"
87
79
  / "pilot-outcome.schema.json")
88
80
  schema = json.loads(schema_path.read_text(encoding="utf-8"))
89
81
  try:
@@ -110,10 +102,13 @@ def register_pilot(project: ProjectWorkspace, *,
110
102
  raise ValueError(
111
103
  f"decision snapshot {decision_snapshot_id} not found in this project; "
112
104
  "a pilot must bind to a real adjudication")
113
- unknown = [o for o in outcome_columns if o not in OUTCOME_TAXONOMY]
105
+ domain = project_domain(project)
106
+ known_tokens = outcome_taxonomy_tokens(domain)
107
+ unknown = [o for o in outcome_columns if o not in known_tokens]
114
108
  if unknown:
115
109
  raise ValueError(
116
- f"outcome_columns outside Outcome Taxonomy: {sorted(unknown)}")
110
+ f"outcome_columns outside the {domain} Outcome Taxonomy: "
111
+ f"{sorted(unknown)}")
117
112
  if not conditions or sample_size < 1:
118
113
  raise ValueError("conditions must be non-empty and sample_size >= 1")
119
114
  if anon_policy.get("no_pii_columns") is not True:
@@ -207,7 +202,13 @@ def link_analysis(project: ProjectWorkspace, pilot_id: str, *,
207
202
  return pilot
208
203
 
209
204
 
210
- def _ensure_outcome(store: GraphStore, outcome_id: str) -> dict:
205
+ def _ensure_outcome(store: GraphStore, outcome_id: str, domain: str) -> dict:
206
+ """Create the pilot outcome row, classified by the DOMAIN registry.
207
+
208
+ ``category_of`` raises for an unregistered token; the caller has already
209
+ validated the token against ``outcome_taxonomy_tokens(domain)``, so an
210
+ error here means the two disagreed and must not be silently absorbed.
211
+ """
211
212
  existing = store.get("outcomes", outcome_id)
212
213
  if existing:
213
214
  return existing
@@ -215,7 +216,7 @@ def _ensure_outcome(store: GraphStore, outcome_id: str) -> dict:
215
216
  return {
216
217
  "outcome_id": outcome_id,
217
218
  "name": token,
218
- "outcome_type": _OUTCOME_CATEGORY.get(token, "learning"),
219
+ "outcome_type": category_of(domain, token),
219
220
  "extensions": {"pilot_outcome": True},
220
221
  }
221
222
 
@@ -236,9 +237,11 @@ def redecide(project: ProjectWorkspace, pilot_id: str, *,
236
237
  raise ValueError(f"invalid effect_direction {effect_direction!r}")
237
238
  if relation_to_claim not in ("support", "contradict", "neutral"):
238
239
  raise ValueError(f"invalid relation_to_claim {relation_to_claim!r}")
239
- if outcome_token not in OUTCOME_TAXONOMY:
240
+ domain = project_domain(project)
241
+ if outcome_token not in outcome_taxonomy_tokens(domain):
240
242
  raise ValueError(
241
- f"outcome_token {outcome_token!r} outside Outcome Taxonomy")
243
+ f"outcome_token {outcome_token!r} outside the {domain} "
244
+ "Outcome Taxonomy")
242
245
  outcome_id = f"OUT-{outcome_token}"
243
246
  pilot = _load_pilot(project, pilot_id)
244
247
  if pilot["status"] not in ("analyzed", "data_imported"):
@@ -281,7 +284,7 @@ def redecide(project: ProjectWorkspace, pilot_id: str, *,
281
284
  "identity_status": "resolved",
282
285
  "extensions": {"pilot_id": pilot_id},
283
286
  }
284
- outcome = _ensure_outcome(store, outcome_id)
287
+ outcome = _ensure_outcome(store, outcome_id, domain)
285
288
  estimate = None
286
289
  if effect_estimate is not None:
287
290
  estimate = {
package/engine/project.py CHANGED
@@ -62,7 +62,7 @@ class ProjectWorkspace:
62
62
 
63
63
  @classmethod
64
64
  def create(cls, home: Path, *, question: str, title: str,
65
- research_mode: str) -> "ProjectWorkspace":
65
+ research_mode: str, domain: str = "education") -> "ProjectWorkspace":
66
66
  home = Path(home).expanduser().resolve()
67
67
  project_id = new_project_id(question)
68
68
  path = home / "projects" / project_id
@@ -72,7 +72,7 @@ class ProjectWorkspace:
72
72
  manifest = {
73
73
  "project_id": project_id,
74
74
  "title": title,
75
- "domain": "education",
75
+ "domain": domain,
76
76
  "question": question,
77
77
  "research_mode": research_mode,
78
78
  "decision_target": _default_decision_target(research_mode),