eduevidence 5.2.0 → 6.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (276) hide show
  1. package/README.md +54 -42
  2. package/README.zh-CN.md +51 -28
  3. package/SKILL.md +390 -133
  4. package/agents/openai.yaml +4 -0
  5. package/assets/readme/controlled-execution.svg +34 -0
  6. package/assets/readme/logo.png +0 -0
  7. package/assets/readme/research-workflow.svg +56 -0
  8. package/assets/readme/studio-graph.png +0 -0
  9. package/assets/readme/studio-overview.png +0 -0
  10. package/assets/readme/studio-reports.png +0 -0
  11. package/autoevolve/config.yaml +17 -0
  12. package/autoevolve/program.md +25 -0
  13. package/autoevolve/protected.manifest.yaml +34 -0
  14. package/benchmarks/adversarial/cases.jsonl +7 -0
  15. package/benchmarks/evidence-library.json +5268 -0
  16. package/benchmarks/partitions.json +8 -0
  17. package/docs/architecture.md +220 -0
  18. package/docs/autoresearch-evolution-plan.md +2903 -0
  19. package/docs/autoresearch-implementation-status.md +101 -0
  20. package/docs/demo-storyboard.md +20 -0
  21. package/docs/demo-workplace-ai.md +92 -0
  22. package/docs/demo.md +32 -0
  23. package/docs/install-guide.md +150 -0
  24. package/docs/orchestration-role-model.md +1254 -0
  25. package/docs/release-closeout/README.md +17 -0
  26. package/docs/release-closeout/frontend-acceptance.md +23 -0
  27. package/docs/release-closeout/issues.md +19 -0
  28. package/docs/release-closeout/verification.md +28 -0
  29. package/docs/release-contract.md +108 -0
  30. package/docs/research-studio-guide.zh-CN.md +166 -0
  31. package/eduevidence_cli.py +17 -11
  32. package/engine/_resources.py +13 -0
  33. package/engine/autoevolve/__init__.py +3 -0
  34. package/engine/autoevolve/agent_view.py +167 -0
  35. package/engine/autoevolve/core.py +357 -0
  36. package/engine/autoevolve/events.py +11 -0
  37. package/engine/autoevolve/git_workspace.py +77 -0
  38. package/engine/autoevolve/projection.py +23 -0
  39. package/engine/autoevolve/runner.py +413 -0
  40. package/engine/autoevolve/trust.py +146 -0
  41. package/engine/autoresearch/__init__.py +6 -0
  42. package/engine/autoresearch/commit.py +132 -0
  43. package/engine/autoresearch/contracts.py +126 -0
  44. package/engine/autoresearch/controller.py +207 -0
  45. package/engine/autoresearch/events.py +12 -0
  46. package/engine/autoresearch/gap_priority.py +168 -0
  47. package/engine/autoresearch/projection.py +30 -0
  48. package/engine/autoresearch/research_memory.py +59 -0
  49. package/engine/autoresearch/saturation.py +91 -0
  50. package/engine/briefs.py +2 -1
  51. package/engine/capabilities.py +1 -0
  52. package/engine/contracts.py +3 -1
  53. package/engine/evidencecore.py +7 -5
  54. package/engine/gaps.py +90 -51
  55. package/engine/judge_pack.py +65 -0
  56. package/engine/library_builtin.py +3 -1
  57. package/engine/living.py +2 -1
  58. package/engine/meta_synthesis.py +3 -1
  59. package/engine/orchestration.py +460 -0
  60. package/engine/pilot.py +2 -1
  61. package/engine/project.py +2 -2
  62. package/engine/research_service.py +113 -0
  63. package/engine/studio_read_model.py +400 -0
  64. package/engine/tribunal.py +1 -2
  65. package/engine/update.py +1 -0
  66. package/engine/versions.py +1 -1
  67. package/engine/worker_result.py +109 -0
  68. package/engine/workflows.py +70 -0
  69. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1720 -0
  70. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
  71. package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
  72. package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
  73. package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
  74. package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
  75. package/examples/ai-coding-assistant-evidence/frame.json +48 -0
  76. package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
  77. package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
  78. package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
  79. package/examples/ai-coding-assistant-evidence/report_spec.json +219 -0
  80. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2614 -0
  81. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2614 -0
  82. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2614 -0
  83. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2614 -0
  84. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2614 -0
  85. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1720 -0
  86. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1720 -0
  87. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1720 -0
  88. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1720 -0
  89. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1720 -0
  90. package/examples/ai-coding-assistant-evidence/result.json +1453 -0
  91. package/examples/ai-coding-assistant-evidence/result.zh.json +1453 -0
  92. package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
  93. package/examples/ai-coding-assistant-evidence/verdict.json +103 -0
  94. package/examples/workplace-ai-assistant/claims.jsonl +4 -0
  95. package/examples/workplace-ai-assistant/evaluation.json +19 -0
  96. package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
  97. package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
  98. package/examples/workplace-ai-assistant/frame.json +41 -0
  99. package/examples/workplace-ai-assistant/intervention.json +27 -0
  100. package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
  101. package/examples/workplace-ai-assistant/methodology.json +60 -0
  102. package/examples/workplace-ai-assistant/report_spec.json +55 -0
  103. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2484 -0
  104. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2484 -0
  105. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2484 -0
  106. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2484 -0
  107. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2484 -0
  108. package/examples/workplace-ai-assistant/result.json +553 -0
  109. package/examples/workplace-ai-assistant/result.zh.json +553 -0
  110. package/examples/workplace-ai-assistant/search_log.json +19 -0
  111. package/examples/workplace-ai-assistant/sources.jsonl +3 -0
  112. package/examples/workplace-ai-assistant/validation_result.json +9 -0
  113. package/examples/workplace-ai-assistant/verdict.json +52 -0
  114. package/install.sh +7 -7
  115. package/integrations/orchestration_dispatch.py +146 -0
  116. package/package.json +37 -3
  117. package/pyproject.toml +11 -20
  118. package/references/autoresearch.md +30 -0
  119. package/references/evaluation-policy.md +24 -0
  120. package/references/orchestration.md +22 -0
  121. package/references/scientific-invariants.md +19 -0
  122. package/retrieval/audit.py +154 -0
  123. package/schemas/intervention.schema.json +106 -21
  124. package/schemas/report-result.schema.json +9 -1
  125. package/schemas/v2/project.schema.json +2 -2
  126. package/schemas/v2/run.schema.json +1 -1
  127. package/schemas/vNext/autoevolve-session.schema.json +1 -0
  128. package/schemas/vNext/eval-snapshot.schema.json +1 -0
  129. package/schemas/vNext/execution-plan.schema.json +1 -0
  130. package/schemas/vNext/gap-priority.schema.json +1 -0
  131. package/schemas/vNext/negative-search-record.schema.json +1 -0
  132. package/schemas/vNext/research-iteration.schema.json +1 -0
  133. package/schemas/vNext/research-strategy.schema.json +1 -0
  134. package/schemas/vNext/skill-experiment.schema.json +1 -0
  135. package/schemas/vNext/task-spec.schema.json +1 -0
  136. package/schemas/vNext/worker-result.schema.json +1 -0
  137. package/scripts/benchmark_judge.py +2 -2
  138. package/scripts/benchmark_v3.py +26 -43
  139. package/scripts/build_esl_artifacts.py +2 -2
  140. package/scripts/build_evidence_library.py +2 -2
  141. package/scripts/build_gh_pages.py +98 -0
  142. package/scripts/build_readme_diagrams.py +72 -0
  143. package/scripts/build_report_variants.py +85 -0
  144. package/scripts/check_autoresearch_invariants.py +95 -0
  145. package/scripts/daily_evolve.py +30 -0
  146. package/scripts/dashboard_server.py +130 -101
  147. package/scripts/did_regression.py +5 -30
  148. package/scripts/enrich_projects_human_and_lieflat.py +1 -1
  149. package/scripts/generate_metrics.py +4 -3
  150. package/scripts/generate_new_projects.py +1 -1
  151. package/scripts/orchestrator.py +172 -18
  152. package/scripts/rebake_all_5themes.py +1 -2
  153. package/scripts/research_auto_cli.py +475 -0
  154. package/scripts/run_workspace.py +17 -7
  155. package/scripts/search_provenance.py +64 -0
  156. package/scripts/serve_web.py +9 -10
  157. package/scripts/skill_lint.py +1 -1
  158. package/scripts/skill_payload.py +78 -0
  159. package/scripts/validate_schema.py +15 -1
  160. package/scripts/vnext_cli.py +133 -0
  161. package/setup.py +12 -0
  162. package/skill/roles/registry.yaml +45 -0
  163. package/skill/sub-skills/report-generation/SKILL.md +12 -6
  164. package/skill/task-briefs/applicability.md +3 -0
  165. package/skill/task-briefs/projection.md +3 -0
  166. package/skill/workflows/decision-and-pilot.md +10 -0
  167. package/skill/workflows/evaluate-and-update.md +10 -0
  168. package/skill/workflows/evidence-review.md +13 -0
  169. package/visualization/eduevidence-report/assets/base.css +2 -2
  170. package/visualization/eduevidence-report/assets/reader.css +752 -0
  171. package/visualization/eduevidence-report/assets/reader.js +132 -0
  172. package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
  173. package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
  174. package/visualization/eduevidence-report/scripts/build_report.py +58 -65
  175. package/visualization/eduevidence-report/scripts/lieflat_engine.py +24 -100
  176. package/visualization/eduevidence-report/themes/academic.css +1 -1
  177. package/visualization/eduevidence-report/themes/claude.css +1 -1
  178. package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
  179. package/visualization/eduevidence-report/themes/datalab.css +2 -2
  180. package/visualization/eduevidence-report/themes/presentation.css +2 -2
  181. package/web/README.md +18 -0
  182. package/web/index.html +53 -0
  183. package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
  184. package/web/studio/assets/index-CzXocaGv.css +1 -0
  185. package/web/studio/assets/index-pa7jD7n4.js +230 -0
  186. package/web/studio/config.json +1 -0
  187. package/web/studio/index.html +14 -0
  188. package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
  189. package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
  190. package/engine/__pycache__/bias.cpython-312.pyc +0 -0
  191. package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
  192. package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
  193. package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
  194. package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
  195. package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
  196. package/engine/__pycache__/events.cpython-312.pyc +0 -0
  197. package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
  198. package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
  199. package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
  200. package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
  201. package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
  202. package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
  203. package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
  204. package/engine/__pycache__/ids.cpython-312.pyc +0 -0
  205. package/engine/__pycache__/library.cpython-312.pyc +0 -0
  206. package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
  207. package/engine/__pycache__/living.cpython-312.pyc +0 -0
  208. package/engine/__pycache__/log.cpython-312.pyc +0 -0
  209. package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
  210. package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
  211. package/engine/__pycache__/migration.cpython-312.pyc +0 -0
  212. package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
  213. package/engine/__pycache__/paths.cpython-312.pyc +0 -0
  214. package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
  215. package/engine/__pycache__/planner.cpython-312.pyc +0 -0
  216. package/engine/__pycache__/project.cpython-312.pyc +0 -0
  217. package/engine/__pycache__/projections.cpython-312.pyc +0 -0
  218. package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
  219. package/engine/__pycache__/run.cpython-312.pyc +0 -0
  220. package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
  221. package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
  222. package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
  223. package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
  224. package/engine/__pycache__/update.cpython-312.pyc +0 -0
  225. package/engine/__pycache__/versions.cpython-312.pyc +0 -0
  226. package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
  227. package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
  228. package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
  229. package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
  230. package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
  231. package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
  232. package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
  233. package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
  234. package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
  235. package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
  236. package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
  237. package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
  238. package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
  239. package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
  240. package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
  241. package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
  242. package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
  243. package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
  244. package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
  245. package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
  246. package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
  247. package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
  248. package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
  249. package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
  250. package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
  251. package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
  252. package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
  253. package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
  254. package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
  255. package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
  256. package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
  257. package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
  258. package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
  259. package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
  260. package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
  261. package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
  262. package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
  263. package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
  264. package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
  265. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
  266. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
  267. package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
  268. package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
  269. package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
  270. package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
  271. package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
  272. package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
  273. package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
  274. package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
  275. package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
  276. package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
@@ -0,0 +1,8 @@
1
+ {
2
+ "version": 1,
3
+ "policy": "Candidate agents may use DEV prompts. HOLDOUT and ADVERSARIAL are evaluator-only inputs during promotion. The legacy questions.jsonl remains for backward compatibility and must not be used as the autoresearch candidate context bundle.",
4
+ "dev": ["Q01","Q02","Q03","Q04","Q05","Q06","Q07","Q08","Q09","Q10","Q11","Q12","Q13","Q14","Q15"],
5
+ "holdout": ["Q16","Q17","Q18","Q19","Q20","Q21","Q22","Q23","Q24","Q25","Q26","Q27","Q28","Q29","Q30"],
6
+ "adversarial": ["ADV-FAKE-DOI","ADV-SNIPPET","ADV-NO-CI","ADV-TASK-LEARNING","ADV-PROMPT-INJECTION","ADV-PII","ADV-SINGULAR-DID"],
7
+ "temporal": {"source": "dynamic", "gold_policy": "timestamped_not_permanent"}
8
+ }
@@ -0,0 +1,220 @@
1
+ # EduEvidence 架构说明
2
+
3
+ EduEvidence 是一个"基于证据的 AI 教学决策与干预"能力(Evidence-Based AI Teaching Decision & Intervention Skill):输入一条教育问题,输出一份可追溯的证据综述、方法论审计、结论判定、试点干预与评估设计。本文档说明其三层架构与双运行模式。
4
+
5
+ ## Canonical Protocol(唯一权威定义)
6
+
7
+ EduEvidence 的端到端流程统一为 **9 步**,由两部分组成。此为本项目唯一权威定义,`docs/methodology.md`、README 等所有文档的协议表述均以本节为准:
8
+
9
+ ```text
10
+ Research Core(6 阶段,证据纪律核心):
11
+ Frame → Retrieve → Extract → Challenge → Audit → Adjudicate
12
+
13
+ Decision Extension(3 阶段,证据到行动):
14
+ Applicability → Intervene → Evaluate
15
+
16
+ 端到端 9 步:
17
+ Frame → Retrieve → Extract → Challenge → Audit → Adjudicate
18
+ → Applicability → Intervene → Evaluate
19
+ ```
20
+
21
+ - `Fetch` 与 `Validate` 是 `Retrieve` 阶段内部的强制 gate(snippet ≠ 证据内容,RULE 2),不单独计为阶段。
22
+ - `Present` 是最终呈现层(报告渲染),不属于协议阶段计数。
23
+ - 每个阶段的输出仍须通过对应 JSON Schema 校验(`schemas/` 顶层共 13 个 Schema,见 §三 目录结构)。
24
+
25
+ ## 〇、EduEvidence Research Engine(V2 内部能力内核)
26
+
27
+ Skill 本体不变;Skill 内部操作 **EduEvidence Research Engine** —— 以 Project/Run/Revision/DecisionSnapshot 为状态模型的持久化研究引擎:
28
+
29
+ - **Project Workspace**(`~/.eduevidence/projects/PRJ-.../`):长期研究项目,持有版本化 **Evidence Graph**(Source→Study→Finding→EvidenceLink→Claim→Outcome→Decision)与 gaps/study-designs/datasets/analyses/decisions/projections/runs。
30
+ - **Evidence Graph 是不可变 revision 模型**:每次提交生成完整快照 `rev-N` 并原子切换 `graph/HEAD`;`result.json`/HTML/Markdown 均为投影,不是事实库。
31
+ - **Shared Research Library**:已验证外部事实(Source/Study/Finding/Audit)经 snapshot import 复用;研究事实可复用,解释(Claim/EvidenceLink/Applicability/Decision)必须项目本地。
32
+ - **两种 Research Mode**:Evidence Review(二手证据)与 Full Research Cycle(证据综述→知识缺口→新研究设计→用户数据→分析→新证据→更新决策)。
33
+ - **冻结科学规则:No new study design without evidence grounding**——任何研究设计必须引用显式、有证据奠基的 KnowledgeGap ID。
34
+ - 引擎是内部能力架构,不是独立服务/应用;Native Core 仅依赖 Python 标准库,不强制 Agent MCP 或 daemon。
35
+
36
+ ## 一、三层架构总览
37
+
38
+ ```
39
+ ┌─────────────────────────────────────────────────────────────────┐
40
+ │ 第 1 层 EduEvidence(领域层) │
41
+ │ 教育领域知识 + 决策 + 干预 + 评价 │
42
+ │ · 教育研究问题结构化(Education Frame) │
43
+ │ · 教学决策(ADOPT / PILOT / REJECT / INSUFFICIENT EVIDENCE) │
44
+ │ · 最小可验证干预(Phase 化试点 + Stop Conditions) │
45
+ │ · 评价设计(Task vs Learning,Immediate vs Retention vs Transfer)│
46
+ ├─────────────────────────────────────────────────────────────────┤
47
+ │ 第 2 层 EvidenceFlow Protocol(流程层) │
48
+ │ Frame → Retrieve → Extract → Challenge → Audit → Adjudicate │
49
+ │ (Retrieve 内含 Fetch/Validate gate;Decision Extension 3 阶段) │
50
+ │ 每一步有独立输入/输出契约,由 13 个 JSON Schema 约束 │
51
+ ├─────────────────────────────────────────────────────────────────┤
52
+ │ 第 3 层 执行层(Execution Layer) │
53
+ │ Mode A Platform Native Mode │
54
+ │ Mode B Agent MCP Enhanced Mode │
55
+ └─────────────────────────────────────────────────────────────────┘
56
+ ```
57
+ ### 1.1 第 1 层:EduEvidence 领域层
58
+
59
+ 领域层承载教育学科的专业判断,不依赖任何具体 Agent 框架:
60
+
61
+ - **教育领域知识**:learner / course / intervention / comparison / outcomes / context 的结构化建模(`education-frame.schema.json`),强制在给出任何教学建议前先完成 Framing。
62
+ - **决策**:基于证据三角的判定输出(`verdict.schema.json`),明确区分"证据支持什么"与"证据不能支持什么",并给出 `adopt / pilot / reject / insufficient_evidence` 四类动作。
63
+ - **干预**:任何决策都默认落到最小可验证的干预设计(`intervention.schema.json`),默认偏向最小可验证 PILOT;只有关键 Outcome 存在较强直接证据、风险可控且场景高度匹配时才允许 ADOPT。Pilot 必须带 Stop Conditions 与 Evidence Alignment。
64
+ - **评价**:为 PILOT / ADOPT 配套评价方案(`evaluation.schema.json`),强制分离 Task Performance 与 Learning Effect,并单独设计 Retention Test 与 Transfer Test。
65
+
66
+ ### 1.2 第 2 层:EvidenceFlow Protocol
67
+
68
+ 证据流协议采用 **Canonical Protocol**(见文首「唯一权威定义」):**Research Core 六阶段**(证据纪律核心)+ **Decision Extension 三阶段**(证据到行动)= **9 步端到端**;`Present` 为最终呈现层,不计入协议阶段。
69
+
70
+ **Research Core(六阶段,可剥离的核心):**
71
+
72
+ | 阶段 | 英文名 | 输入 | 输出 |
73
+ |------|--------|------|------|
74
+ | 1. 框定 | Frame | 原始教育问题 | Education Research Frame(含 decision_target、scope、inclusion/exclusion criteria) |
75
+ | 2. 检索 | Retrieve | Frame | 校验通过的来源与 Claim 级证据基础(内部强制 gate:Fetch 抓取全文 + Validate 来源/内容校验,snippet ≠ 证据内容,RULE 2;对应 `source.schema.json` / `fetch-result.schema.json`) |
76
+ | 3. 抽取 | Extract | 校验后的来源文献 | Claim 级证据对象(`evidence.schema.json`) |
77
+ | 4. 质询 | Challenge | 证据对象 | 反方证据、负面结果、未发现、confounder 清单 |
78
+ | 5. 审计 | Audit | 证据对象 | 方法学审计(`methodology.schema.json`),含 task_vs_learning_guard |
79
+ | 6. 裁决 | Adjudicate | 全部证据 + 审计 | Education Verdict + Recommended Action + Confidence |
80
+
81
+ **Decision Extension(三阶段,证据到行动,即端到端第 7–9 步):**
82
+
83
+ | 阶段 | 英文名 | 输入 | 输出 |
84
+ |------|--------|------|------|
85
+ | 7. 适用 | Applicability | Verdict | 适用性分析(For whom / which course / which outcome / what conditions) |
86
+ | 8. 干预 | Intervene | Verdict + Applicability | Teaching Intervention(最小可验证试点 + 停止条件) |
87
+ | 9. 评价 | Evaluate | 干预方案 | Evaluation Plan(基线/后测/保持/迁移 + 成功阈值) |
88
+
89
+ **Present(呈现,不计入 9 步协议):**
90
+
91
+ | 阶段 | 英文名 | 输入 | 输出 |
92
+ |------|--------|------|------|
93
+ | 10. 呈现 | Present | result.json + result.zh.json | 单文件双语 HTML 报告 + 信息图 + 学术图 + **AI 自由组合的 Lieflat 数据驱动画廊**(`visualization/`;主题在生成前从 `claude`[Light] / `academic`[Light] / `datalab`[Light] / `datalab-dark`[Dark] / `presentation`[Dark] 五选一,最终 HTML 不提供主题切换,仅保留中英文切换) |
94
+
95
+ **Present 可视化管线(AI 组合 + 数据驱动):**
96
+
97
+ ```text
98
+ result.json.visual_layout(AI 写图表计划:type + 目录编号 + 双语文案 + 数据源参数)
99
+ │ resolve_visual_layout(build_report.py)
100
+ │ · 只接受注册表内 type(lieflat_engine.REGISTRY),未注册显式报错
101
+ │ · 双语缺失 / 参数非法 → 丢弃该条 + 原因入 report_spec
102
+ │ · 缺失或全无效 → 确定性安全组合(forest + dot_cascade +
103
+ │ bubble_almanac + tick_rows)
104
+ ▼
105
+ charts_data.py 提取器(唯一数字来源,读 result.json → 规范化 bundle)
106
+ │ 数据不足 → 抑制该图 + 原因(镜像 Meaningful Visualization Gate)
107
+ ▼
108
+ lieflat_engine.render_figure(type, bundle, theme, meta)
109
+ │ 主题化内联 SVG:lf-pop/lf-fade/lf-draw + --motion-delay stagger,
110
+ │ 无内嵌 <style>、无硬编码演示数据;每个显示数值登记 audit
111
+ ▼
112
+ 完整性门 lieflat_data_bound(compute_integrity + 页脚)
113
+ │ 渲染值逐一比对提取器 bundle——篡改数值天然不被采用
114
+ ▼
115
+ motion/motion.css + motion/motion.js(data-lieflat reveal:滚入播放、
116
+ 点击重播 + timer 清理、prefers-reduced-motion 降级、打印全开)
117
+ ```
118
+
119
+ 学术图(outcome-comparison / benchmark / forest)保持主题无关的出版级渲染;Lieflat 部分只渲染 `resolve_visual_layout` 校验通过的条目。注册表、契约与推荐组合见 `visualization/eduevidence-report/references/lieflat-composition.md`,schema 见 `visualization/eduevidence-report/schemas/visual-layout.schema.json`。
120
+
121
+ 该协议是**可剥离的**:即使没有 Agent 框架,只要按此协议组织检索、抽取、审计与裁决,也能得到可复现的决策链。
122
+
123
+ ### 1.3 第 3 层:执行层
124
+
125
+ 执行层负责把第 1、2 层的能力实际跑起来,提供两种运行模式。
126
+
127
+ ## 二、双运行模式
128
+
129
+ ### Mode A:Platform Native Mode(平台原生模式)
130
+
131
+ - **不依赖** daemon、CLI、Agent MCP 或任何外部进程,是纯提示词/静态技能交付。
132
+ - **SKILL.md 可单独理解**:单文件包含领域知识、EvidenceFlow Protocol、JSON Schema 说明、示例与复现命令,任一支持自定义指令的 AI 平台均可直接加载。
133
+ - 适合:Claude Projects、ChatGPT Custom Instructions、Cursor Rules 等"以文档为技能载体"的平台。
134
+ - 局限:检索依赖模型自身或平台内建工具,上下文不持久,跨会话一致性靠提示词保证。
135
+
136
+ ### Mode B:Agent MCP Enhanced Mode(Agent MCP 增强模式)
137
+
138
+ 在 Mode A 基础上,通过 MCP(Model Context Protocol)接入执行工具,获得以下增强能力:
139
+
140
+ - **多 CLI**:可同时挂载检索、代码、测试、Schema 校验等命令行工具,检索不再是模型"猜测来源",而是真实调用检索器。
141
+ - **多模型**:允许检索/抽取/裁决由不同模型承担,例如轻量模型负责 Retrieve,强推理模型负责 Adjudicate。
142
+ - **独立上下文**:每个子任务拥有独立 Context,避免长对话稀释关键证据,裁决阶段再合并。
143
+ - **超时恢复**:子任务可设置超时与重试,网络/工具失败可降级回 Mode A 语义继续。
144
+ - **成本优化**:按阶段选择模型与 Token 预算,Framing 与 Extraction 用低成本路径,裁决与审计用高成本路径。
145
+ - **Memory Bank**:跨会话缓存已审来源、历史裁决与常用检索词,二次提问命中缓存则跳过重复检索。
146
+
147
+ | 能力维度 | Mode A Platform Native | Mode B Agent MCP Enhanced |
148
+ |----------|------------------------|---------------------------|
149
+ | 依赖 | 无(纯 SKILL.md) | daemon / CLI / MCP Server |
150
+ | 检索 | 模型内建/平台工具 | 真实多 CLI 检索 |
151
+ | 模型 | 单模型 | 多模型分工 |
152
+ | 上下文 | 单上下文 | 独立子上下文 + 汇总 |
153
+ | 超时 | 无 | 可配置超时与恢复 |
154
+ | 成本 | 固定 | 分阶段 Token 预算优化 |
155
+ | 记忆 | 无 | Memory Bank 跨会话缓存 |
156
+
157
+ **推荐组合**:文档与协议以 Mode A 形式交付(任何人可独立理解与使用),接入 MCP 后自动升级为 Mode B(检索更真实、决策更稳健、成本更低)。
158
+
159
+ ## 三、项目目录结构
160
+
161
+ ```
162
+ edu/
163
+ ├── SKILL.md # 技能入口:EduEvidence 使用说明(Mode A 可独立理解)
164
+ ├── README.md / README.zh-CN.md # 双语说明(英文 / 中文)
165
+ ├── pyproject.toml # 打包元数据(wheel 自带 CLI + engine;核心零第三方依赖)
166
+ ├── install.sh # 一键安装(本地 / 多 Agent Skill)+ 自检
167
+ ├── skill/ # 技能组件
168
+ │ ├── agents/ # 8 个角色协议(education-planner / evidence-retriever /
169
+ │ │ # evidence-analyst / skeptic / method-reviewer /
170
+ │ │ # evidence-judge / intervention-designer / evaluation-designer)
171
+ │ ├── sub-skills/ # 12 个子技能 SKILL.md(原 skills/,v5 合并至此)
172
+ │ └── task-briefs/ # 任务简报模板
173
+ ├── references/ # 15 份教育方法论文档(education-framing / outcome-taxonomy /
174
+ │ # evidence-quality / methodology-audit / skeptic-protocol /
175
+ │ # tribunal-policy / applicability-policy / intervention-design /
176
+ │ # evaluation-design / retrieval-protocol / source-validity)
177
+ ├── schemas/ # 13 个顶层 JSON Schema 数据契约 + schemas/v2/(V2 契约)
178
+ │ ├── education-frame.schema.json
179
+ │ ├── source.schema.json
180
+ │ ├── fetch-result.schema.json
181
+ │ ├── evidence.schema.json
182
+ │ ├── cross-model-review.schema.json
183
+ │ ├── methodology.schema.json
184
+ │ ├── verdict.schema.json
185
+ │ ├── intervention.schema.json
186
+ │ ├── evaluation.schema.json
187
+ │ ├── report-result.schema.json
188
+ │ ├── report-spec.schema.json
189
+ │ ├── chart-spec.schema.json
190
+ │ └── agent-mcp-approval.schema.json
191
+ ├── engine/ # Research Engine 内核(Project / Run / Revision /
192
+ │ # DecisionSnapshot、Evidence Graph、tribunal / synthesis /
193
+ │ # gaps / study-design / datasets / analysis / projections、
194
+ │ # v3: pilot 决策闭环 / meta_synthesis 跨项目综述)
195
+ ├── scripts/ # 工具脚本(validate_schema / pre_verdict_gate /
196
+ │ # compute_confidence / orchestrator / benchmark /
197
+ │ # render_report 等确定性逻辑)
198
+ ├── retrieval/ # 检索与抓取层(fetch / validate / dedupe / failures)
199
+ ├── integrations/ # 集成层(Agent MCP 增强 + Smart Web Fetch)
200
+ ├── visualization/ # 呈现层(eduevidence-report:build_report / build_charts /
201
+ │ # build_infographics / build_figures / charts_data 提取器 /
202
+ │ # lieflat_engine 注册表渲染器 / motion / themes / schemas;
203
+ │ # lieflat-charts:图表品味法典正本)
204
+ ├── benchmarks/ # 基准评测(questions / annotations / baselines /
205
+ │ # evaluator / results / v2)
206
+ ├── examples/ # 端到端示例(按教育问题组织)
207
+ │ └── ai-coding-assistant/ # 主 Demo:大一 C 语言课程 AI 编程助手
208
+ ├── docs/ # 本文档(架构/方法学/Benchmark/Demo/复现指南)
209
+ └── tests/ # pytest 测试
210
+ ```
211
+
212
+ 各目录职责单一:`schemas/` 是协议契约,`scripts/` 是校验与评测入口,`benchmarks/` 沉淀评估资产,`examples/` 提供可复现案例,`docs/` 解释"为什么这样设计"。
213
+
214
+ ## 四、设计原则
215
+
216
+ 1. **领域层独立于 Agent 层**:EduEvidence 的判断逻辑不耦合任何框架,未来换 Agent 方案时第 1、2 层无需改动。
217
+ 2. **协议可剥离**:EvidenceFlow Protocol 可脱离 MCP 单独执行,保证最低运行成本可用。
218
+ 3. **决策可追溯**:从 Verdict 出发可反查到 evidence_id、source_location、方法学审计项与置信度分解。
219
+ 4. **默认偏向最小可验证 PILOT**:领域层内置"先试点、后部署"约束,从架构上防止跳过验证直接给出全量方案;只有关键 Outcome 存在较强直接证据、风险可控且场景高度匹配时才允许 ADOPT。
220
+ 5. **降级友好**:Mode B 的任何子任务失败,都可降级为 Mode A 语义继续产出,保证结果不中断。