eduevidence 5.2.0 → 6.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (276) hide show
  1. package/README.md +54 -42
  2. package/README.zh-CN.md +51 -28
  3. package/SKILL.md +390 -133
  4. package/agents/openai.yaml +4 -0
  5. package/assets/readme/controlled-execution.svg +34 -0
  6. package/assets/readme/logo.png +0 -0
  7. package/assets/readme/research-workflow.svg +56 -0
  8. package/assets/readme/studio-graph.png +0 -0
  9. package/assets/readme/studio-overview.png +0 -0
  10. package/assets/readme/studio-reports.png +0 -0
  11. package/autoevolve/config.yaml +17 -0
  12. package/autoevolve/program.md +25 -0
  13. package/autoevolve/protected.manifest.yaml +34 -0
  14. package/benchmarks/adversarial/cases.jsonl +7 -0
  15. package/benchmarks/evidence-library.json +5268 -0
  16. package/benchmarks/partitions.json +8 -0
  17. package/docs/architecture.md +220 -0
  18. package/docs/autoresearch-evolution-plan.md +2903 -0
  19. package/docs/autoresearch-implementation-status.md +101 -0
  20. package/docs/demo-storyboard.md +20 -0
  21. package/docs/demo-workplace-ai.md +92 -0
  22. package/docs/demo.md +32 -0
  23. package/docs/install-guide.md +150 -0
  24. package/docs/orchestration-role-model.md +1254 -0
  25. package/docs/release-closeout/README.md +17 -0
  26. package/docs/release-closeout/frontend-acceptance.md +23 -0
  27. package/docs/release-closeout/issues.md +19 -0
  28. package/docs/release-closeout/verification.md +28 -0
  29. package/docs/release-contract.md +108 -0
  30. package/docs/research-studio-guide.zh-CN.md +166 -0
  31. package/eduevidence_cli.py +17 -11
  32. package/engine/_resources.py +13 -0
  33. package/engine/autoevolve/__init__.py +3 -0
  34. package/engine/autoevolve/agent_view.py +167 -0
  35. package/engine/autoevolve/core.py +357 -0
  36. package/engine/autoevolve/events.py +11 -0
  37. package/engine/autoevolve/git_workspace.py +77 -0
  38. package/engine/autoevolve/projection.py +23 -0
  39. package/engine/autoevolve/runner.py +413 -0
  40. package/engine/autoevolve/trust.py +146 -0
  41. package/engine/autoresearch/__init__.py +6 -0
  42. package/engine/autoresearch/commit.py +132 -0
  43. package/engine/autoresearch/contracts.py +126 -0
  44. package/engine/autoresearch/controller.py +207 -0
  45. package/engine/autoresearch/events.py +12 -0
  46. package/engine/autoresearch/gap_priority.py +168 -0
  47. package/engine/autoresearch/projection.py +30 -0
  48. package/engine/autoresearch/research_memory.py +59 -0
  49. package/engine/autoresearch/saturation.py +91 -0
  50. package/engine/briefs.py +2 -1
  51. package/engine/capabilities.py +1 -0
  52. package/engine/contracts.py +3 -1
  53. package/engine/evidencecore.py +7 -5
  54. package/engine/gaps.py +90 -51
  55. package/engine/judge_pack.py +65 -0
  56. package/engine/library_builtin.py +3 -1
  57. package/engine/living.py +2 -1
  58. package/engine/meta_synthesis.py +3 -1
  59. package/engine/orchestration.py +460 -0
  60. package/engine/pilot.py +2 -1
  61. package/engine/project.py +2 -2
  62. package/engine/research_service.py +113 -0
  63. package/engine/studio_read_model.py +400 -0
  64. package/engine/tribunal.py +1 -2
  65. package/engine/update.py +1 -0
  66. package/engine/versions.py +1 -1
  67. package/engine/worker_result.py +109 -0
  68. package/engine/workflows.py +70 -0
  69. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1720 -0
  70. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
  71. package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
  72. package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
  73. package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
  74. package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
  75. package/examples/ai-coding-assistant-evidence/frame.json +48 -0
  76. package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
  77. package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
  78. package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
  79. package/examples/ai-coding-assistant-evidence/report_spec.json +219 -0
  80. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2614 -0
  81. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2614 -0
  82. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2614 -0
  83. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2614 -0
  84. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2614 -0
  85. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1720 -0
  86. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1720 -0
  87. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1720 -0
  88. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1720 -0
  89. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1720 -0
  90. package/examples/ai-coding-assistant-evidence/result.json +1453 -0
  91. package/examples/ai-coding-assistant-evidence/result.zh.json +1453 -0
  92. package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
  93. package/examples/ai-coding-assistant-evidence/verdict.json +103 -0
  94. package/examples/workplace-ai-assistant/claims.jsonl +4 -0
  95. package/examples/workplace-ai-assistant/evaluation.json +19 -0
  96. package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
  97. package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
  98. package/examples/workplace-ai-assistant/frame.json +41 -0
  99. package/examples/workplace-ai-assistant/intervention.json +27 -0
  100. package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
  101. package/examples/workplace-ai-assistant/methodology.json +60 -0
  102. package/examples/workplace-ai-assistant/report_spec.json +55 -0
  103. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2484 -0
  104. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2484 -0
  105. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2484 -0
  106. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2484 -0
  107. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2484 -0
  108. package/examples/workplace-ai-assistant/result.json +553 -0
  109. package/examples/workplace-ai-assistant/result.zh.json +553 -0
  110. package/examples/workplace-ai-assistant/search_log.json +19 -0
  111. package/examples/workplace-ai-assistant/sources.jsonl +3 -0
  112. package/examples/workplace-ai-assistant/validation_result.json +9 -0
  113. package/examples/workplace-ai-assistant/verdict.json +52 -0
  114. package/install.sh +7 -7
  115. package/integrations/orchestration_dispatch.py +146 -0
  116. package/package.json +37 -3
  117. package/pyproject.toml +11 -20
  118. package/references/autoresearch.md +30 -0
  119. package/references/evaluation-policy.md +24 -0
  120. package/references/orchestration.md +22 -0
  121. package/references/scientific-invariants.md +19 -0
  122. package/retrieval/audit.py +154 -0
  123. package/schemas/intervention.schema.json +106 -21
  124. package/schemas/report-result.schema.json +9 -1
  125. package/schemas/v2/project.schema.json +2 -2
  126. package/schemas/v2/run.schema.json +1 -1
  127. package/schemas/vNext/autoevolve-session.schema.json +1 -0
  128. package/schemas/vNext/eval-snapshot.schema.json +1 -0
  129. package/schemas/vNext/execution-plan.schema.json +1 -0
  130. package/schemas/vNext/gap-priority.schema.json +1 -0
  131. package/schemas/vNext/negative-search-record.schema.json +1 -0
  132. package/schemas/vNext/research-iteration.schema.json +1 -0
  133. package/schemas/vNext/research-strategy.schema.json +1 -0
  134. package/schemas/vNext/skill-experiment.schema.json +1 -0
  135. package/schemas/vNext/task-spec.schema.json +1 -0
  136. package/schemas/vNext/worker-result.schema.json +1 -0
  137. package/scripts/benchmark_judge.py +2 -2
  138. package/scripts/benchmark_v3.py +26 -43
  139. package/scripts/build_esl_artifacts.py +2 -2
  140. package/scripts/build_evidence_library.py +2 -2
  141. package/scripts/build_gh_pages.py +98 -0
  142. package/scripts/build_readme_diagrams.py +72 -0
  143. package/scripts/build_report_variants.py +85 -0
  144. package/scripts/check_autoresearch_invariants.py +95 -0
  145. package/scripts/daily_evolve.py +30 -0
  146. package/scripts/dashboard_server.py +130 -101
  147. package/scripts/did_regression.py +5 -30
  148. package/scripts/enrich_projects_human_and_lieflat.py +1 -1
  149. package/scripts/generate_metrics.py +4 -3
  150. package/scripts/generate_new_projects.py +1 -1
  151. package/scripts/orchestrator.py +172 -18
  152. package/scripts/rebake_all_5themes.py +1 -2
  153. package/scripts/research_auto_cli.py +475 -0
  154. package/scripts/run_workspace.py +17 -7
  155. package/scripts/search_provenance.py +64 -0
  156. package/scripts/serve_web.py +9 -10
  157. package/scripts/skill_lint.py +1 -1
  158. package/scripts/skill_payload.py +78 -0
  159. package/scripts/validate_schema.py +15 -1
  160. package/scripts/vnext_cli.py +133 -0
  161. package/setup.py +12 -0
  162. package/skill/roles/registry.yaml +45 -0
  163. package/skill/sub-skills/report-generation/SKILL.md +12 -6
  164. package/skill/task-briefs/applicability.md +3 -0
  165. package/skill/task-briefs/projection.md +3 -0
  166. package/skill/workflows/decision-and-pilot.md +10 -0
  167. package/skill/workflows/evaluate-and-update.md +10 -0
  168. package/skill/workflows/evidence-review.md +13 -0
  169. package/visualization/eduevidence-report/assets/base.css +2 -2
  170. package/visualization/eduevidence-report/assets/reader.css +752 -0
  171. package/visualization/eduevidence-report/assets/reader.js +132 -0
  172. package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
  173. package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
  174. package/visualization/eduevidence-report/scripts/build_report.py +58 -65
  175. package/visualization/eduevidence-report/scripts/lieflat_engine.py +24 -100
  176. package/visualization/eduevidence-report/themes/academic.css +1 -1
  177. package/visualization/eduevidence-report/themes/claude.css +1 -1
  178. package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
  179. package/visualization/eduevidence-report/themes/datalab.css +2 -2
  180. package/visualization/eduevidence-report/themes/presentation.css +2 -2
  181. package/web/README.md +18 -0
  182. package/web/index.html +53 -0
  183. package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
  184. package/web/studio/assets/index-CzXocaGv.css +1 -0
  185. package/web/studio/assets/index-pa7jD7n4.js +230 -0
  186. package/web/studio/config.json +1 -0
  187. package/web/studio/index.html +14 -0
  188. package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
  189. package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
  190. package/engine/__pycache__/bias.cpython-312.pyc +0 -0
  191. package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
  192. package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
  193. package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
  194. package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
  195. package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
  196. package/engine/__pycache__/events.cpython-312.pyc +0 -0
  197. package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
  198. package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
  199. package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
  200. package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
  201. package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
  202. package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
  203. package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
  204. package/engine/__pycache__/ids.cpython-312.pyc +0 -0
  205. package/engine/__pycache__/library.cpython-312.pyc +0 -0
  206. package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
  207. package/engine/__pycache__/living.cpython-312.pyc +0 -0
  208. package/engine/__pycache__/log.cpython-312.pyc +0 -0
  209. package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
  210. package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
  211. package/engine/__pycache__/migration.cpython-312.pyc +0 -0
  212. package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
  213. package/engine/__pycache__/paths.cpython-312.pyc +0 -0
  214. package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
  215. package/engine/__pycache__/planner.cpython-312.pyc +0 -0
  216. package/engine/__pycache__/project.cpython-312.pyc +0 -0
  217. package/engine/__pycache__/projections.cpython-312.pyc +0 -0
  218. package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
  219. package/engine/__pycache__/run.cpython-312.pyc +0 -0
  220. package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
  221. package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
  222. package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
  223. package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
  224. package/engine/__pycache__/update.cpython-312.pyc +0 -0
  225. package/engine/__pycache__/versions.cpython-312.pyc +0 -0
  226. package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
  227. package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
  228. package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
  229. package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
  230. package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
  231. package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
  232. package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
  233. package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
  234. package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
  235. package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
  236. package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
  237. package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
  238. package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
  239. package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
  240. package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
  241. package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
  242. package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
  243. package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
  244. package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
  245. package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
  246. package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
  247. package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
  248. package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
  249. package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
  250. package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
  251. package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
  252. package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
  253. package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
  254. package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
  255. package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
  256. package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
  257. package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
  258. package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
  259. package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
  260. package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
  261. package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
  262. package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
  263. package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
  264. package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
  265. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
  266. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
  267. package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
  268. package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
  269. package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
  270. package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
  271. package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
  272. package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
  273. package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
  274. package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
  275. package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
  276. package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
@@ -0,0 +1,78 @@
1
+ """Shared, explicit runtime allowlist for flat Skills, npm installs and wheels."""
2
+ from __future__ import annotations
3
+
4
+ from pathlib import Path
5
+ import shutil
6
+ import sys
7
+
8
+ TREES = (
9
+ "agents", "engine", "domains", "skill", "references", "schemas", "scripts",
10
+ "retrieval", "integrations", "visualization/eduevidence-report", "web/studio",
11
+ "assets/readme",
12
+ )
13
+ FILES = (
14
+ "SKILL.md", "eduevidence_cli.py", "install.sh", "pyproject.toml", "setup.py",
15
+ "LICENSE", "CHANGELOG.md", "README.md", "README.zh-CN.md", "web/index.html",
16
+ "benchmarks/evidence-library.json", "benchmarks/partitions.json",
17
+ "benchmarks/adversarial/cases.jsonl",
18
+ )
19
+ DOCS = (
20
+ "architecture.md", "demo.md", "demo-storyboard.md", "install-guide.md",
21
+ "reproducibility.md", "release-contract.md", "autoresearch-evolution-plan.md",
22
+ "orchestration-role-model.md", "autoresearch-implementation-status.md",
23
+ "research-studio-guide.zh-CN.md", "demo-workplace-ai.md",
24
+ "release-closeout/README.md", "release-closeout/issues.md",
25
+ "release-closeout/frontend-acceptance.md", "release-closeout/verification.md",
26
+ )
27
+ EXAMPLES = (
28
+ "ai-coding-assistant-evidence", "workplace-ai-assistant",
29
+ )
30
+ EXAMPLE_FILES = (
31
+ "result.json", "result.zh.json", "evidence_graph.json", "report_spec.json",
32
+ "EduEvidence_Report.html", "verdict.json", "evidence.jsonl", "sources.jsonl",
33
+ "frame.json", "methodology.json", "intervention.json", "evaluation.json",
34
+ )
35
+ RETIRED_DEMO_SCRIPTS = {
36
+ "scripts/build_esl_artifacts.py", "scripts/generate_new_projects.py",
37
+ "scripts/enrich_projects_human_and_lieflat.py", "scripts/build_killer_demo.py",
38
+ "scripts/sync_killer_demo_report.py",
39
+ }
40
+
41
+
42
+ def payload_files(root: Path):
43
+ """Yield safe relative files; no symlinks, private state or generated caches."""
44
+ paths = set(FILES)
45
+ paths.update(f"docs/{name}" for name in DOCS)
46
+ for tree in TREES:
47
+ paths.update(p.relative_to(root).as_posix() for p in (root / tree).rglob("*") if p.is_file())
48
+ # Ship configuration, never historical results or private Autoevolve sessions.
49
+ paths.update(f"autoevolve/{name}" for name in ("program.md", "config.yaml", "protected.manifest.yaml"))
50
+ for example in EXAMPLES:
51
+ paths.update(f"examples/{example}/{name}" for name in EXAMPLE_FILES)
52
+ paths.update(p.relative_to(root).as_posix() for p in (root / "examples" / example / "reports-5themes").glob("*.html"))
53
+ for relative in sorted(paths):
54
+ if relative in RETIRED_DEMO_SCRIPTS:
55
+ continue
56
+ path = root / relative
57
+ parts = Path(relative).parts
58
+ if any(p.startswith(".") or p in {"__pycache__", "node_modules", "runs", "venv", "test-results"} for p in parts):
59
+ continue
60
+ if path.suffix in {".pyc", ".pyo", ".log"} or any((root.joinpath(*parts[:i])).is_symlink() for i in range(1, len(parts) + 1)):
61
+ continue
62
+ if path.is_file():
63
+ yield relative
64
+
65
+
66
+ def copy_payload(root: Path, destination: Path) -> None:
67
+ required = ("SKILL.md", "agents/openai.yaml", "web/studio/index.html", "eduevidence_cli.py")
68
+ for name in required:
69
+ if not (root / name).is_file():
70
+ raise FileNotFoundError(f"Incomplete Skill runtime: {name}")
71
+ for relative in payload_files(root):
72
+ output = destination / relative
73
+ output.parent.mkdir(parents=True, exist_ok=True)
74
+ shutil.copy2(root / relative, output)
75
+
76
+
77
+ if __name__ == "__main__":
78
+ copy_payload(Path(sys.argv[1]).resolve(), Path(sys.argv[2]).resolve())
@@ -121,7 +121,21 @@ class Validator:
121
121
  """Validate `value` against `schema` (draft-07 subset). Raises SchemaError."""
122
122
  if "$ref" in schema:
123
123
  # draft-07: $ref replaces sibling keywords entirely
124
- self.validate(value, self._resolve_ref(schema["$ref"], path), path)
124
+ ref = schema["$ref"]
125
+ if not ref.startswith("#") and self.base_dir is not None:
126
+ filename, _, fragment = ref.partition("#")
127
+ target = self.base_dir / filename
128
+ if not target.is_file():
129
+ raise SchemaError(f"{path}: unresolvable $ref {ref!r}")
130
+ cache_key = str(target.resolve())
131
+ if cache_key not in self._ref_cache:
132
+ self._ref_cache[cache_key] = json.loads(target.read_text(encoding="utf-8"))
133
+ document = self._ref_cache[cache_key]
134
+ validator = Validator(document, base_dir=target.parent)
135
+ referenced = validator._resolve_ref("#" + fragment, path) if fragment else document
136
+ validator.validate(value, referenced, path)
137
+ else:
138
+ self.validate(value, self._resolve_ref(ref, path), path)
125
139
  return
126
140
 
127
141
  if "type" in schema:
@@ -0,0 +1,133 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ import json
5
+ from pathlib import Path
6
+
7
+ try:
8
+ from research_auto_cli import research_auto
9
+ except ImportError: # imported as scripts.vnext_cli in tests/package contexts
10
+ from scripts.research_auto_cli import research_auto
11
+
12
+
13
+ def _read_json(path, default=None):
14
+ if not path:
15
+ return default
16
+ return json.loads(Path(path).read_text(encoding="utf-8"))
17
+
18
+
19
+ def evolve(argv):
20
+ parser = argparse.ArgumentParser(prog="eduevidence evolve")
21
+ sub = parser.add_subparsers(dest="action", required=True)
22
+ for name in ("init", "status", "report", "best"):
23
+ cmd = sub.add_parser(name)
24
+ cmd.add_argument("--root", default=".")
25
+ cmd = sub.add_parser("baseline")
26
+ cmd.add_argument("--root", default=".")
27
+ cmd.add_argument("--eval", required=True)
28
+ cmd = sub.add_parser("run")
29
+ cmd.add_argument("--root", default=".")
30
+ cmd.add_argument("--experiment", required=True)
31
+ cmd.add_argument("--baseline-eval", required=True)
32
+ cmd.add_argument("--candidate-eval", required=True)
33
+ cmd = sub.add_parser("prepare-pr")
34
+ cmd.add_argument("--root", default=".")
35
+ args = parser.parse_args(argv)
36
+ repo = Path(args.root).resolve()
37
+ root = repo / "autoevolve"
38
+ root.mkdir(parents=True, exist_ok=True)
39
+ from engine.autoevolve import (
40
+ DailyProfile,
41
+ EvalSnapshot,
42
+ ExperimentLog,
43
+ PlateauTracker,
44
+ ProtectedManifest,
45
+ SkillExperiment,
46
+ promote,
47
+ )
48
+ if args.action == "init":
49
+ DailyProfile().validate()
50
+ (root / "runs").mkdir(exist_ok=True)
51
+ if not (root / "best.json").exists():
52
+ (root / "best.json").write_text('{"best_experiment_id": null}\n', encoding="utf-8")
53
+ ExperimentLog(root)
54
+ print(root)
55
+ return 0
56
+ if args.action in {"status", "report"}:
57
+ rows = (
58
+ (root / "results.tsv").read_text(encoding="utf-8").splitlines()[1:]
59
+ if (root / "results.tsv").exists()
60
+ else []
61
+ )
62
+ best = _read_json(root / "best.json", {}) or {}
63
+ statuses = [row.split("\t")[-2] for row in rows if "\t" in row]
64
+ print(
65
+ json.dumps(
66
+ {
67
+ "experiments": len(rows),
68
+ "best": best,
69
+ "plateau": PlateauTracker().plateau(statuses),
70
+ },
71
+ indent=2,
72
+ )
73
+ )
74
+ return 0
75
+ if args.action == "best":
76
+ print(json.dumps(_read_json(root / "best.json", {}), indent=2))
77
+ return 0
78
+ if args.action == "baseline":
79
+ data = _read_json(args.eval)
80
+ (root / "baseline.json").write_text(
81
+ json.dumps(data, indent=2) + "\n", encoding="utf-8"
82
+ )
83
+ print(data.get("eval_id", "baseline"))
84
+ return 0
85
+ if args.action == "run":
86
+ experiment = SkillExperiment(**_read_json(args.experiment))
87
+ baseline = EvalSnapshot(**_read_json(args.baseline_eval))
88
+ candidate = EvalSnapshot(**_read_json(args.candidate_eval))
89
+ manifest = ProtectedManifest.from_repo(repo)
90
+ ok, bad = manifest.validate_changes(experiment.changed_files)
91
+ scope_ok, scope_bad = manifest.validate_mutation_scope(
92
+ experiment.changed_files,
93
+ mutation_tiers=experiment.mutation_scope,
94
+ allow_controlled="controlled" in experiment.mutation_scope,
95
+ )
96
+ if not ok:
97
+ status, reason = "INVALID", "protected mutation: " + ",".join(bad)
98
+ elif not scope_ok:
99
+ status, reason = "INVALID", "mutation outside approved tier: " + ",".join(scope_bad)
100
+ else:
101
+ status, reason = promote(baseline, candidate)
102
+ experiment.status = status
103
+ experiment.promotion_reason = reason
104
+ ExperimentLog(root).append(experiment, candidate=candidate, description=reason)
105
+ if status == "KEEP":
106
+ (root / "best.json").write_text(
107
+ json.dumps(
108
+ {
109
+ "best_experiment_id": experiment.experiment_id,
110
+ "candidate_commit": experiment.candidate_commit,
111
+ "eval_id": candidate.eval_id,
112
+ },
113
+ indent=2,
114
+ )
115
+ + "\n",
116
+ encoding="utf-8",
117
+ )
118
+ print(json.dumps({"status": status, "reason": reason}, indent=2))
119
+ return 0
120
+ if args.action == "prepare-pr":
121
+ best = _read_json(root / "best.json", {}) or {}
122
+ print(
123
+ json.dumps(
124
+ {
125
+ "promotion": "branch_only",
126
+ "best": best,
127
+ "note": "Human approval is required to open/merge the final PR.",
128
+ },
129
+ indent=2,
130
+ )
131
+ )
132
+ return 0
133
+ return 2
package/setup.py ADDED
@@ -0,0 +1,12 @@
1
+ """Include the same runtime resources as the flat Skill in installed wheels."""
2
+ from pathlib import Path
3
+ import runpy
4
+ from setuptools import setup
5
+
6
+ root = Path(__file__).resolve().parent
7
+ payload_files = runpy.run_path(str(root / "scripts" / "skill_payload.py"))["payload_files"]
8
+ groups = {}
9
+ for relative in payload_files(root):
10
+ destination = str(Path("share/eduevidence") / Path(relative).parent)
11
+ groups.setdefault(destination, []).append(relative)
12
+ setup(data_files=sorted(groups.items()))
@@ -0,0 +1,45 @@
1
+ roles:
2
+ education-planner:
3
+ responsibility: framing completeness and scope
4
+ stages: [frame]
5
+ capabilities: [research-planning]
6
+ critical_path: true
7
+ evidence-retriever:
8
+ responsibility: source acquisition and provenance
9
+ stages: [retrieve]
10
+ capabilities: [literature-review, full-text-fetch, source-validation]
11
+ evidence-analyst:
12
+ responsibility: structured finding extraction
13
+ stages: [extract]
14
+ capabilities: [evidence-extraction]
15
+ skeptic:
16
+ responsibility: independent counter-evidence coverage
17
+ stages: [challenge]
18
+ capabilities: [contradiction-analysis, counter-retrieval]
19
+ independence_required: true
20
+ critical_path: true
21
+ method-reviewer:
22
+ responsibility: methodology and construct-validity appraisal
23
+ stages: [audit]
24
+ capabilities: [methodology-audit]
25
+ independence_required: true
26
+ critical_path: true
27
+ evidence-judge:
28
+ responsibility: evidence-bounded adjudication and applicability
29
+ stages: [adjudicate, applicability]
30
+ capabilities: [evidence-review, decision-adjudication, applicability]
31
+ critical_path: true
32
+ intervention-designer:
33
+ responsibility: grounded intervention or pilot design
34
+ stages: [intervene]
35
+ capabilities: [study-design, intervention-design]
36
+ evaluation-designer:
37
+ responsibility: estimable evaluation and update logic
38
+ stages: [evaluate]
39
+ capabilities: [evaluation-design, data-analysis]
40
+
41
+ execution:
42
+ max_parallel_workers: 6
43
+ split_by: evidence_axis
44
+ single_writer: lead-orchestrator
45
+ recursive_swarm: false
@@ -7,23 +7,29 @@ description: "Renders 5 baked-theme single-file bilingual HTML reports, executiv
7
7
  ## 1. When to Use
8
8
  Trigger at the completion of a research cycle (Step 9: Present stage) to deliver visual dossiers, decision briefs, and publication-grade reports.
9
9
 
10
- ## 2. Lieflat Charts Editorial Standards Integration
11
- All statistical figures, evidence matrices, and causal trajectories follow the **Lieflat Charts Editorial Codex** (`visualization/lieflat-charts/`). Chart galleries are **AI-composed but data-driven**: the upstream AI writes a chart plan (`visual_layout`), and the deterministic renderer extracts every number from `result.json` via `scripts/charts_data.py`. **The AI never writes numeric values into the layout** — values are un-tamperable and always traceable.
10
+ ## 2. AI-Composed Data Visualization
11
+ Use the self-contained selection catalog at `visualization/eduevidence-report/references/chart-selection-catalog.md` together with the executable registry in `references/lieflat-composition.md`. Chart galleries are **AI-composed but data-driven**: the upstream AI writes a chart plan (`visual_layout`), and the deterministic renderer extracts every number from `result.json` via `scripts/charts_data.py`. **The AI never writes numeric values into the layout** — values are traceable and cannot be replaced by invented chart data.
12
12
 
13
13
  ### 2.1 Six-Step Chart Composition Workflow
14
14
  1. **判数据形状。** 看 `result.json` 的数据长什么样(效应量 g+CI / 年份×维度 / 方向计数 / 阶段周区间 / 置信度单值 / 审计状态……),形状是选图的主键。
15
- 2. **按 catalog 审计候选。** 在 `visualization/lieflat-charts/catalog.md` 按数据形状召回候选,至少比较 3 个并写下淘汰理由(语义契合、单位诚实、标签容纳、阅读速度、本批次是否重复)。Glance 系只在 Lupi/Basics 不适配或用户明确要求快读时进入候选。
15
+ 2. **按自包含 catalog 审计候选。** 在 `visualization/eduevidence-report/references/chart-selection-catalog.md` 按数据形状召回候选;通常比较至少 3 个可行图型并写下淘汰理由(语义契合、单位诚实、标签容纳、阅读速度、移动端适配、本批次是否重复)。如果数据形状只有 1–2 个有效候选,不得为了凑满 3 个选择语义错误的图。
16
16
  3. **锁定注册表 `type`。** 只能选 `visualization/eduevidence-report/references/lieflat-composition.md` 注册表内的图型(type ↔ 目录编号 ↔ 数据形状 ↔ 提取器);**未注册 type 会显式报错并被丢弃,不存在静默回退**。
17
17
  4. **写 `report_outline` + `visual_layout`。** 每张图承担一个独立结论;总数 ≤6 张;同一批形状不重复(不堆同类环/条/点阵);每张图写清 `type / catalog_ref / title_zh+en / subtitle_zh+en / caption_zh+en / source / params`,副标题写清图例与单位;主题色系由烘焙主题锁定,布局不得换色。
18
18
  5. **渲染。** 运行 `build_report.py`(或 `scripts/rebake_all_5themes.py`)——渲染器对每个条目走注册表提取器:数据不足 → 该图抑制并记录原因(镜像 Meaningful Visualization Gate);全部无效 → 确定性安全组合(forest_plot + dot_cascade + bubble_almanac + tick_rows)。
19
- 6. **按 Lieflat skill 第八节自检**(面积 sqrt、最小字号 6.5/5.5px、数值 800、reveal 滚入播放 + 点击重播 + reduced-motion、卡片四件套齐全、数值与视觉成正比)。
19
+ 6. **按本地 catalog + 渲染不变量自检**(面积 sqrt、最小字号 6.5/5.5px、数值 800、reveal 滚入播放 + 点击重播 + reduced-motion、卡片四件套齐全、数值与视觉成正比)。
20
20
 
21
21
  ### 2.2 数据契约:AI 写计划,渲染器出数
22
22
  - `visual_layout` 条目 = 图型 + 目录编号 + 双语文案 + 数据源参数。**数值一律由 `scripts/charts_data.py` 的提取器从 `result.json` 读出**,渲染器只接收提取器 bundle——被篡改的数值天然不被采用,完整性门 `lieflat_data_bound` 逐值核对溯源。
23
23
  - 50 篇级大样本 → 提取器 top-N 截断 + SVG `<title>` 悬停读数,不加欺骗性交互。
24
24
  - 中文长类目 → 按决策树选横排图(F5/F1/F6 系),L2 cascade 类目名 ≤4 字约束保留。
25
25
 
26
- ### 2.3 5 Theme Palettes Adaptive Binding (Zero-CDN Guarantee)
26
+ ### 2.3 Tables are audit surfaces, not failed visualizations
27
+ - Keep `EvidenceMatrix`, `SourceList`, methodology details, and `ClaimTrace` for exact row-level verification and provenance.
28
+ - Use charts for patterns, comparisons, distributions, composition, time/phase structure, and decision summaries.
29
+ - Never replace a provenance table merely to make the report look more visual. A chart may summarize a table, but the auditable rows remain reachable.
30
+ - If no chart adds information without inventing missing values or mixing incompatible outcomes, suppress the visualization and explain why.
31
+
32
+ ### 2.4 5 Theme Palettes Adaptive Binding (Zero-CDN Guarantee)
27
33
  Every chart is rendered as **pure self-contained inline SVG** following the baked report theme's color system:
28
34
  - **`claude` (智库典雅)** ➔ **Lieflat Palm (暖棕)**: `#FAF7F2` paper base, `#B8694A` terracotta, `#5E8A6A` forest green, `#C99A4A` amber.
29
35
  - **`academic` (学术顶刊)** ➔ **Lieflat Mono / Nature (学术黑白灰)**: `#FFFFFF` paper base, `#0F172A` ink black, Okabe-Ito colorblind-safe accents.
@@ -33,7 +39,7 @@ Every chart is rendered as **pure self-contained inline SVG** following the bake
33
39
 
34
40
  Dark themes draw on the theme's `card_bg`; text contrast is checked against the theme palette (≥4.5:1 for body/labels). The chart structure is identical across light/dark — only colors change.
35
41
 
36
- ### 2.4 排版守则(五主题统一约束)
42
+ ### 2.5 排版守则(五主题统一约束)
37
43
  五个烘焙主题的排版必须通过 `scripts/lint_report_layout.py`(静态不变量 + 浏览器级
38
44
  390/768/1280 × brief/full 实测):轨道 `minmax(0,1fr)` / `minmax(min(Npx,100%),1fr)`、
39
45
  主题自带移动端媒体覆写、禁止裸 `1fr` 与固定 px 最小值 auto-fit、表格外包 `overflow-x:auto`。
@@ -0,0 +1,3 @@
1
+ # Applicability stage
2
+
3
+ Assess transportability to the target population and setting. State supported population, implementation conditions, excluded populations, outcome limits, and uncertainty. Do not infer applicability from the presence of a positive result.
@@ -0,0 +1,3 @@
1
+ # Projection stage
2
+
3
+ Build reports and exports only from completed artifacts and decision snapshots. Projection errors must not change a scientific result. Record the exact source revision and rendered artifact hash.
@@ -0,0 +1,10 @@
1
+ ---
2
+ name: decision-and-pilot
3
+ description: Convert a bounded evidence decision into a reversible, measurable pilot.
4
+ ---
5
+
6
+ # Decision & Pilot
7
+
8
+ Use only after an Evidence Review has produced an auditable decision snapshot. Add `Intervene` with a minimal pilot, explicit stop conditions, owner, population, and outcome measures. A pilot is not an adoption claim.
9
+
10
+ When this workflow is reached from Evidence Autoresearch, require the bridge in `references/autoresearch.md`: the KnowledgeGap is HIGH-DVI and decision-material, remains unresolved, bounded secondary search is saturated, and the empirical study is ethically/operationally feasible. Then still pass the existing grounded StudyDesign gate; “few papers found” alone never authorizes a pilot.
@@ -0,0 +1,10 @@
1
+ ---
2
+ name: evaluate-and-update
3
+ description: Re-inject validated outcome data into an evidence graph and re-adjudicate.
4
+ ---
5
+
6
+ # Evaluate & Update
7
+
8
+ Use when pilot or field data exists. Validate provenance and missingness before analysis, fail closed when inference is not estimable, commit a graph revision, then produce a new decision snapshot and a decision diff.
9
+
10
+ For autonomous/living refreshes, preserve prior revisions and treat the newly re-adjudicated decision as a candidate update until the applicable review/human gate accepts it. New evidence does not need to flip the action; unchanged action with changed certainty, applicability, or boundary is a valid revision.
@@ -0,0 +1,13 @@
1
+ ---
2
+ name: evidence-review
3
+ description: Traceable evidence review with mandatory counter-evidence and methodology gates.
4
+ ---
5
+
6
+ # Evidence Review
7
+
8
+ Use for a question that needs a bounded, decision-grade evidence assessment.
9
+ Run `Frame → Retrieve → Extract → Challenge → Audit → Adjudicate → Applicability`.
10
+
11
+ The required outputs are a search plan and attempt log, validated sources, claim-level evidence links, a methodology audit, a decision boundary, and applicability limits. Search snippets are discovery metadata, never evidence.
12
+
13
+ When the user asks to continue autonomously, identify the next most decision-relevant evidence, or keep iterating until the evidence state reaches a bounded stopping condition, load `references/autoresearch.md`. Keep the public workflow unchanged: Evidence Autoresearch is a meta-layer over this review, not a fourth user-facing workflow. Preserve append-only evidence and the Single Writer rule.
@@ -17,7 +17,7 @@ body { margin:0; background:var(--bg); color:var(--text);
17
17
  .controls { width:calc(100% - 36px); max-width:1200px; margin:0 auto 16px; padding:12px clamp(0px,1vw,12px); display:flex; gap:18px; flex-wrap:wrap; align-items:center;
18
18
  border-bottom:1px solid var(--border); }
19
19
  .report-header { border-bottom:1px solid var(--border); padding-bottom:16px; margin-bottom:24px; }
20
- .report-header h1 { font-family:var(--font-head); font-size:1.9rem; margin:0 0 8px; color:var(--text); }
20
+ .report-header h1 { font-family:var(--font-head); font-size:1.55rem; margin:0 0 8px; color:var(--text); }
21
21
  .report-header .meta { color:var(--insufficient); font-size:.85rem; }
22
22
  .lang-switcher { display:flex; gap:8px; align-items:center; flex-wrap:wrap; }
23
23
  .lang-switcher span { font-size:.82rem; color:var(--insufficient); }
@@ -300,7 +300,7 @@ button:focus-visible, a:focus-visible, input:focus-visible, select:focus-visible
300
300
  .controls { width:calc(100% - 24px); margin-bottom:10px; padding-left:8px; padding-right:8px; }
301
301
  .report-shell { width:100%; padding:12px 14px 56px; }
302
302
  .report-header { width:100%; }
303
- .report-header h1 { font-size:1.55rem; overflow-wrap:anywhere; }
303
+ .report-header h1 { font-size:1.3rem; overflow-wrap:anywhere; }
304
304
  .report-header .meta, .brief-source h3, .claim-cell, .detail-body, .source-detail-grid dd { overflow-wrap:anywhere; word-break:break-word; }
305
305
  .data-table { font-size:.78rem; }
306
306
  .hero-insights, .tribunal-grid, .scope-grid, .retrieval-grid, .brief-source-grid, .method-audit-grid { grid-template-columns:1fr; }