eduevidence 5.2.0 → 6.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (386) hide show
  1. package/CONTRIBUTING.md +105 -0
  2. package/README.md +142 -75
  3. package/README.zh-CN.md +73 -30
  4. package/SKILL.md +397 -131
  5. package/agents/openai.yaml +4 -0
  6. package/assets/readme/controlled-execution.svg +34 -0
  7. package/assets/readme/landing-tour.gif +0 -0
  8. package/assets/readme/logo.png +0 -0
  9. package/assets/readme/research-workflow.svg +56 -0
  10. package/assets/readme/studio-graph.png +0 -0
  11. package/assets/readme/studio-overview.png +0 -0
  12. package/assets/readme/studio-reports.png +0 -0
  13. package/assets/readme/studio-tour.gif +0 -0
  14. package/autoevolve/config.yaml +17 -0
  15. package/autoevolve/program.md +25 -0
  16. package/autoevolve/protected.manifest.yaml +34 -0
  17. package/benchmarks/adversarial/cases.jsonl +7 -0
  18. package/benchmarks/evidence-library.json +5268 -0
  19. package/benchmarks/partitions.json +8 -0
  20. package/bin/eduevidence.js +2 -1
  21. package/docs/architecture.md +496 -0
  22. package/docs/autoresearch-evolution-plan.md +2903 -0
  23. package/docs/autoresearch-implementation-status.md +101 -0
  24. package/docs/demo-storyboard.md +20 -0
  25. package/docs/demo-workplace-ai.md +92 -0
  26. package/docs/demo.md +32 -0
  27. package/docs/install-guide.md +150 -0
  28. package/docs/orchestration-role-model.md +1254 -0
  29. package/docs/release-closeout/README.md +17 -0
  30. package/docs/release-closeout/frontend-acceptance.md +23 -0
  31. package/docs/release-closeout/issues.md +19 -0
  32. package/docs/release-closeout/verification.md +28 -0
  33. package/docs/release-contract.md +108 -0
  34. package/docs/research-studio-guide.zh-CN.md +166 -0
  35. package/docs/sciverse-api.md +125 -0
  36. package/eduevidence_cli.py +29 -13
  37. package/engine/_resources.py +13 -0
  38. package/engine/autoevolve/__init__.py +3 -0
  39. package/engine/autoevolve/agent_view.py +167 -0
  40. package/engine/autoevolve/core.py +357 -0
  41. package/engine/autoevolve/events.py +11 -0
  42. package/engine/autoevolve/git_workspace.py +77 -0
  43. package/engine/autoevolve/projection.py +23 -0
  44. package/engine/autoevolve/runner.py +413 -0
  45. package/engine/autoevolve/trust.py +146 -0
  46. package/engine/autoresearch/__init__.py +6 -0
  47. package/engine/autoresearch/commit.py +132 -0
  48. package/engine/autoresearch/contracts.py +126 -0
  49. package/engine/autoresearch/controller.py +207 -0
  50. package/engine/autoresearch/events.py +12 -0
  51. package/engine/autoresearch/gap_priority.py +168 -0
  52. package/engine/autoresearch/projection.py +30 -0
  53. package/engine/autoresearch/research_memory.py +59 -0
  54. package/engine/autoresearch/saturation.py +91 -0
  55. package/engine/briefs.py +2 -1
  56. package/engine/capabilities.py +1 -0
  57. package/engine/contracts.py +3 -1
  58. package/engine/decision_policy.py +96 -0
  59. package/engine/evidence_graph.py +14 -10
  60. package/engine/evidencecore.py +7 -5
  61. package/engine/gaps.py +132 -73
  62. package/engine/ids.py +2 -0
  63. package/engine/judge_pack.py +65 -0
  64. package/engine/library.py +6 -2
  65. package/engine/library_builtin.py +3 -1
  66. package/engine/living.py +36 -5
  67. package/engine/meta_synthesis.py +3 -1
  68. package/engine/migration.py +88 -3
  69. package/engine/orchestration.py +460 -0
  70. package/engine/paths.py +2 -0
  71. package/engine/pilot.py +36 -33
  72. package/engine/project.py +2 -2
  73. package/engine/research_service.py +113 -0
  74. package/engine/studio_read_model.py +400 -0
  75. package/engine/taxonomy.py +211 -0
  76. package/engine/tribunal.py +44 -33
  77. package/engine/update.py +1 -0
  78. package/engine/versions.py +1 -1
  79. package/engine/worker_result.py +109 -0
  80. package/engine/workflows.py +70 -0
  81. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +2934 -0
  82. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
  83. package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
  84. package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
  85. package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
  86. package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
  87. package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
  88. package/examples/ai-coding-assistant-evidence/frame.json +48 -0
  89. package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
  90. package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
  91. package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
  92. package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
  93. package/examples/ai-coding-assistant-evidence/report_spec.json +230 -0
  94. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2934 -0
  95. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2934 -0
  96. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2934 -0
  97. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2934 -0
  98. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2934 -0
  99. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +2934 -0
  100. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +2934 -0
  101. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +2934 -0
  102. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +2934 -0
  103. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +2934 -0
  104. package/examples/ai-coding-assistant-evidence/result.json +1457 -0
  105. package/examples/ai-coding-assistant-evidence/result.zh.json +1457 -0
  106. package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
  107. package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
  108. package/examples/ai-coding-assistant-evidence/verdict.json +107 -0
  109. package/examples/spaced-retrieval-practice/applicability.json +14 -0
  110. package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
  111. package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
  112. package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
  113. package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
  114. package/examples/spaced-retrieval-practice/frame.json +58 -0
  115. package/examples/spaced-retrieval-practice/gate_report.json +101 -0
  116. package/examples/spaced-retrieval-practice/methodology.json +78 -0
  117. package/examples/spaced-retrieval-practice/report_spec.json +212 -0
  118. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
  119. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
  120. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
  121. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
  122. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
  123. package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
  124. package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
  125. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
  126. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
  127. package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
  128. package/examples/spaced-retrieval-practice/result.json +942 -0
  129. package/examples/spaced-retrieval-practice/result.zh.json +942 -0
  130. package/examples/spaced-retrieval-practice/skeptic.json +70 -0
  131. package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
  132. package/examples/spaced-retrieval-practice/verdict.json +93 -0
  133. package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
  134. package/examples/workplace-ai-assistant/claims.jsonl +4 -0
  135. package/examples/workplace-ai-assistant/evaluation.json +19 -0
  136. package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
  137. package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
  138. package/examples/workplace-ai-assistant/final_verdict.json +78 -0
  139. package/examples/workplace-ai-assistant/frame.json +41 -0
  140. package/examples/workplace-ai-assistant/gate_report.json +101 -0
  141. package/examples/workplace-ai-assistant/intervention.json +27 -0
  142. package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
  143. package/examples/workplace-ai-assistant/methodology.json +60 -0
  144. package/examples/workplace-ai-assistant/report_spec.json +224 -0
  145. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2814 -0
  146. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2814 -0
  147. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2814 -0
  148. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2814 -0
  149. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2814 -0
  150. package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
  151. package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
  152. package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
  153. package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
  154. package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
  155. package/examples/workplace-ai-assistant/result.json +615 -0
  156. package/examples/workplace-ai-assistant/result.zh.json +615 -0
  157. package/examples/workplace-ai-assistant/search_log.json +19 -0
  158. package/examples/workplace-ai-assistant/skeptic.json +72 -0
  159. package/examples/workplace-ai-assistant/sources.jsonl +3 -0
  160. package/examples/workplace-ai-assistant/validation_result.json +9 -0
  161. package/examples/workplace-ai-assistant/verdict.json +78 -0
  162. package/install.sh +7 -7
  163. package/integrations/agent_mcp.py +2 -2
  164. package/integrations/orchestration_dispatch.py +146 -0
  165. package/package.json +46 -3
  166. package/pyproject.toml +14 -22
  167. package/references/autoresearch.md +30 -0
  168. package/references/evaluation-policy.md +24 -0
  169. package/references/orchestration.md +22 -0
  170. package/references/report-copy-style.md +67 -0
  171. package/references/retrieval-compliance.md +75 -0
  172. package/references/retrieval-protocol.md +20 -0
  173. package/references/scientific-invariants.md +19 -0
  174. package/retrieval/audit.py +178 -0
  175. package/retrieval/fetch.py +96 -0
  176. package/retrieval/sciverse.py +398 -0
  177. package/retrieval/search.py +47 -7
  178. package/schemas/applicability.schema.json +94 -0
  179. package/schemas/chart-spec.schema.json +10 -3
  180. package/schemas/evidence.schema.json +316 -43
  181. package/schemas/fetch-result.schema.json +2 -1
  182. package/schemas/intervention.schema.json +106 -21
  183. package/schemas/report-result.schema.json +12 -4
  184. package/schemas/report-spec.schema.json +98 -100
  185. package/schemas/skeptic.schema.json +86 -0
  186. package/schemas/source.schema.json +21 -2
  187. package/schemas/v2/finding.schema.json +5 -1
  188. package/schemas/v2/methodology-audit.schema.json +5 -1
  189. package/schemas/v2/outcome.schema.json +28 -5
  190. package/schemas/v2/project.schema.json +2 -2
  191. package/schemas/v2/run.schema.json +1 -1
  192. package/schemas/v2/study.schema.json +5 -1
  193. package/schemas/vNext/autoevolve-session.schema.json +34 -0
  194. package/schemas/vNext/eval-snapshot.schema.json +77 -0
  195. package/schemas/vNext/execution-plan.schema.json +50 -0
  196. package/schemas/vNext/gap-priority.schema.json +54 -0
  197. package/schemas/vNext/negative-search-record.schema.json +68 -0
  198. package/schemas/vNext/research-iteration.schema.json +87 -0
  199. package/schemas/vNext/research-strategy.schema.json +62 -0
  200. package/schemas/vNext/skill-experiment.schema.json +90 -0
  201. package/schemas/vNext/task-spec.schema.json +156 -0
  202. package/schemas/vNext/worker-result.schema.json +60 -0
  203. package/schemas/verdict.schema.json +164 -28
  204. package/scripts/benchmark_judge.py +2 -2
  205. package/scripts/benchmark_v3.py +26 -43
  206. package/scripts/build_esl_artifacts.py +4 -4
  207. package/scripts/build_evidence_library.py +2 -2
  208. package/scripts/build_gh_pages.py +98 -0
  209. package/scripts/build_readme_diagrams.py +72 -0
  210. package/scripts/build_report_variants.py +101 -0
  211. package/scripts/build_result.py +74 -9
  212. package/scripts/check_autoresearch_invariants.py +95 -0
  213. package/scripts/check_package_parity.py +85 -0
  214. package/scripts/check_protocol_alignment.py +375 -0
  215. package/scripts/check_versioned_schemas.py +254 -0
  216. package/scripts/claim_audit.py +13 -8
  217. package/scripts/compute_confidence.py +10 -0
  218. package/scripts/daily_evolve.py +30 -0
  219. package/scripts/dashboard_server.py +130 -101
  220. package/scripts/did_regression.py +17 -32
  221. package/scripts/enrich_projects_human_and_lieflat.py +1 -1
  222. package/scripts/evidence_score.py +5 -2
  223. package/scripts/generate_metrics.py +4 -3
  224. package/scripts/generate_new_projects.py +5 -5
  225. package/scripts/orchestrator.py +286 -36
  226. package/scripts/pre_verdict_gate.py +224 -26
  227. package/scripts/quickstart.py +18 -2
  228. package/scripts/rebake_all_5themes.py +1 -2
  229. package/scripts/research_auto_cli.py +475 -0
  230. package/scripts/run_workspace.py +24 -8
  231. package/scripts/search_provenance.py +64 -0
  232. package/scripts/serve_web.py +9 -10
  233. package/scripts/skill_lint.py +1 -1
  234. package/scripts/skill_payload.py +81 -0
  235. package/scripts/test_adversarial_empirical.py +26 -19
  236. package/scripts/validate_schema.py +46 -2
  237. package/scripts/vnext_cli.py +133 -0
  238. package/setup.py +12 -0
  239. package/skill/agents/evaluation-designer.md +20 -4
  240. package/skill/agents/evidence-analyst.md +19 -3
  241. package/skill/agents/evidence-judge.md +50 -2
  242. package/skill/agents/evidence-retriever.md +20 -3
  243. package/skill/agents/intervention-designer.md +20 -4
  244. package/skill/agents/method-reviewer.md +18 -2
  245. package/skill/agents/{education-planner.md → research-planner.md} +19 -3
  246. package/skill/agents/skeptic.md +18 -2
  247. package/skill/roles/registry.yaml +45 -0
  248. package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
  249. package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
  250. package/skill/sub-skills/data-analysis/SKILL.md +34 -15
  251. package/skill/sub-skills/ethics-review/SKILL.md +33 -10
  252. package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
  253. package/skill/sub-skills/evidence-review/SKILL.md +31 -12
  254. package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
  255. package/skill/sub-skills/literature-review/SKILL.md +35 -14
  256. package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
  257. package/skill/sub-skills/report-generation/SKILL.md +40 -6
  258. package/skill/sub-skills/research-planning/SKILL.md +41 -14
  259. package/skill/sub-skills/study-design/SKILL.md +30 -9
  260. package/skill/task-briefs/adjudicate.md +32 -7
  261. package/skill/task-briefs/applicability.md +38 -0
  262. package/skill/task-briefs/audit.md +32 -7
  263. package/skill/task-briefs/challenge.md +34 -5
  264. package/skill/task-briefs/evaluate.md +30 -5
  265. package/skill/task-briefs/extract.md +31 -8
  266. package/skill/task-briefs/frame.md +39 -10
  267. package/skill/task-briefs/intervene.md +32 -6
  268. package/skill/task-briefs/present.md +32 -8
  269. package/skill/task-briefs/projection.md +37 -0
  270. package/skill/task-briefs/retrieve.md +36 -6
  271. package/skill/workflows/decision-and-pilot.md +85 -0
  272. package/skill/workflows/evaluate-and-update.md +93 -0
  273. package/skill/workflows/evidence-review.md +117 -0
  274. package/visualization/eduevidence-report/assets/base.css +2 -2
  275. package/visualization/eduevidence-report/assets/reader.css +752 -0
  276. package/visualization/eduevidence-report/assets/reader.js +132 -0
  277. package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
  278. package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
  279. package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
  280. package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
  281. package/visualization/eduevidence-report/scripts/build_report.py +561 -121
  282. package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
  283. package/visualization/eduevidence-report/scripts/lieflat_engine.py +371 -136
  284. package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
  285. package/visualization/eduevidence-report/themes/academic.css +1 -1
  286. package/visualization/eduevidence-report/themes/claude.css +1 -1
  287. package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
  288. package/visualization/eduevidence-report/themes/datalab.css +2 -2
  289. package/visualization/eduevidence-report/themes/presentation.css +2 -2
  290. package/web/README.md +18 -0
  291. package/web/architecture.html +14885 -0
  292. package/web/index.html +53 -0
  293. package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
  294. package/web/studio/assets/index-B8tkF44Q.css +1 -0
  295. package/web/studio/assets/index-CQ6Keoyc.js +230 -0
  296. package/web/studio/config.json +1 -0
  297. package/web/studio/index.html +14 -0
  298. package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
  299. package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
  300. package/engine/__pycache__/bias.cpython-312.pyc +0 -0
  301. package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
  302. package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
  303. package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
  304. package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
  305. package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
  306. package/engine/__pycache__/events.cpython-312.pyc +0 -0
  307. package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
  308. package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
  309. package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
  310. package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
  311. package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
  312. package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
  313. package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
  314. package/engine/__pycache__/ids.cpython-312.pyc +0 -0
  315. package/engine/__pycache__/library.cpython-312.pyc +0 -0
  316. package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
  317. package/engine/__pycache__/living.cpython-312.pyc +0 -0
  318. package/engine/__pycache__/log.cpython-312.pyc +0 -0
  319. package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
  320. package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
  321. package/engine/__pycache__/migration.cpython-312.pyc +0 -0
  322. package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
  323. package/engine/__pycache__/paths.cpython-312.pyc +0 -0
  324. package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
  325. package/engine/__pycache__/planner.cpython-312.pyc +0 -0
  326. package/engine/__pycache__/project.cpython-312.pyc +0 -0
  327. package/engine/__pycache__/projections.cpython-312.pyc +0 -0
  328. package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
  329. package/engine/__pycache__/run.cpython-312.pyc +0 -0
  330. package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
  331. package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
  332. package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
  333. package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
  334. package/engine/__pycache__/update.cpython-312.pyc +0 -0
  335. package/engine/__pycache__/versions.cpython-312.pyc +0 -0
  336. package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
  337. package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
  338. package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
  339. package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
  340. package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
  341. package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
  342. package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
  343. package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
  344. package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
  345. package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
  346. package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
  347. package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
  348. package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
  349. package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
  350. package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
  351. package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
  352. package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
  353. package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
  354. package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
  355. package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
  356. package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
  357. package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
  358. package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
  359. package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
  360. package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
  361. package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
  362. package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
  363. package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
  364. package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
  365. package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
  366. package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
  367. package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
  368. package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
  369. package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
  370. package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
  371. package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
  372. package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
  373. package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
  374. package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
  375. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
  376. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
  377. package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
  378. package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
  379. package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
  380. package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
  381. package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
  382. package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
  383. package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
  384. package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
  385. package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
  386. package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
@@ -0,0 +1,86 @@
1
+ {
2
+ "decision_question": "大一 C 语言课程是否应该允许学生使用生成式 AI 编程助手?",
3
+ "target_population": "university first-year computer science students learning C programming for the first time",
4
+ "target_context": "16-week lecture-lab course, 60 students, TA support, offline",
5
+ "supported_claims": [
6
+ "AI coding assistants reliably increase task performance during training (completion speed, correctness) — E-001, E-006.",
7
+ "Unguarded generative AI access can harm independent problem solving when access is removed — E-004.",
8
+ "Guardrail design (hints instead of answers) substantially mitigates the negative learning effect — E-005.",
9
+ "Task performance gains do not automatically imply learning gains — E-004 vs E-006 (within-study contrast).",
10
+ "Tool capability is substantial: Codex solves roughly half to three-quarters of CS1 exam-style questions — E-010.",
11
+ "Professional-developer RCT shows ~55% faster task completion with Copilot; directness limited by professional population — E-008.",
12
+ "LLM code explanations rate comparable to student-authored explanations, viable as scaffold material — E-011."
13
+ ],
14
+ "uncertain_claims": [
15
+ "Whether AI coding assistants improve or preserve actual programming learning in university novices — no direct university-level RCT in reviewed set [无直接证据]",
16
+ "Whether one-week neutral retention (Kazemitabaar 2023) extends to a semester — E-003.",
17
+ "Whether benchmark quality findings (E-009) and explanation-quality ratings (E-011) translate into classroom learning gains.",
18
+ "How comprehension/ownership difficulties documented in usability studies (E-012) behave over a full semester with guardrails."
19
+ ],
20
+ "contradicted_claims": [
21
+ "The claim 'AI tools always improve learning' is contradicted by E-004 (unguarded access, -17% independent exam).",
22
+ "The claim 'speed gains equal learning gains' is contradicted by the task-vs-learning separation across E-001/E-006/E-008 vs E-004."
23
+ ],
24
+ "reason_for_disagreement": "Disagreement comes from outcome separation (task vs learning), tool design (guarded vs unguarded), and population (K-12 / professionals vs university). Task-performance evidence is consistently positive across randomized and benchmark studies; the only study measuring independent performance after AI removal shows harm without guardrails; usability and artifact studies add dependence and quality caveats rather than resolving the learning question.",
25
+ "methodology_summary": "Eight real sources: three randomized experiments (Kazemitabaar 2023 n=69 K-12; Bastani 2025 n≈950 high-school mathematics; Peng 2023 n=95 professional developers, preprint), one ESL writing mixed-methods study (Marzuki 2024), plus benchmark/capability/usability studies (Yetistiren 2023; Finnie-Ansley 2022; explanation-comparison 2023; Vaithilingam 2022). No direct RCT in university programming courses. Internal validity of the core RCTs is strong; directness to first-year university C programming is weak. All sources carry registry-verified DOIs (see benchmarks/doi-audit/report.md).",
26
+ "outcome_specific_findings": {
27
+ "completion_time": "positive during training and professional tasks (E-001, E-008)",
28
+ "independent_problem_solving": "neutral-to-negative without guardrails (E-002, E-004)",
29
+ "retention": "neutral over 1 week (E-003)",
30
+ "assignment_score": "positive during practice, negative on closed-book exam (E-004, E-006); tool itself scores passing-level on CS1 questions (E-010)",
31
+ "code_quality": "mixed on benchmarks; security concerns documented (E-009)",
32
+ "metacognition": "LLM explanations compare well (E-011) while novice ownership/debugging difficulties persist (E-012)",
33
+ "ai_dependency": "documented crutch behavior with unguarded tool (E-004, E-005, E-012)"
34
+ },
35
+ "short_term_effect": "Task performance reliably increases; learning effect null-to-negative without guardrails.",
36
+ "long_term_effect": "No evidence beyond one week; long-term learning effect unknown.",
37
+ "transfer_effect": "No full transfer evidence; manual code-modification not harmed in one small study (E-002).",
38
+ "risk_effect": "AI dependency and over-reliance risk is real and documented for unguarded usage (E-004) and foreshadowed by usability findings (E-012).",
39
+ "applicability": {
40
+ "suitable_for": "pilot in first-year C course with guardrailed usage policy",
41
+ "not_suitable_for": "unrestricted AI adoption without usage policy",
42
+ "required_conditions": [
43
+ "guardrailed AI usage policy (hints not answers, modeled on GPT Tutor arm)",
44
+ "no-AI transfer assessment",
45
+ "TA support"
46
+ ]
47
+ },
48
+ "confidence": "Moderate",
49
+ "confidence_breakdown": {
50
+ "score": 0.5,
51
+ "evidence_quality": 0.7,
52
+ "consistency": 0.6,
53
+ "directness": 0.4,
54
+ "evidence_count": 12,
55
+ "independent_studies": 8,
56
+ "independent_samples": 8,
57
+ "count_term": 1.0,
58
+ "conflict_penalty": 0.0,
59
+ "unsupported_penalty": 0.0
60
+ },
61
+ "what_can_be_claimed": [
62
+ "AI coding assistants raise task performance for novices during training.",
63
+ "Unguarded access carries a real risk of hurting independent problem solving.",
64
+ "Guardrail design can mitigate that risk.",
65
+ "Direct evidence for university C programming learning is missing.",
66
+ "Tool capability headroom is large (CS1 question pass rates; professional speed RCT)."
67
+ ],
68
+ "what_cannot_be_claimed": [
69
+ "AI coding assistants improve (or even preserve) university students' programming learning.",
70
+ "Any long-term or retention benefit.",
71
+ "Any claim about which students benefit, based on university samples.",
72
+ "That benchmark or usability findings substitute for classroom learning outcomes."
73
+ ],
74
+ "missing_evidence": [
75
+ "RCT of AI coding assistants in university programming courses with retention and no-AI transfer tests.",
76
+ "Studies varying AI usage policy within the same course.",
77
+ "Longitudinal data on AI dependency beyond one course.",
78
+ "Peer-reviewed replication of the professional speed RCT (Peng et al. remains a preprint)."
79
+ ],
80
+ "recommended_action": "pilot",
81
+ "decision_rationale": "Positive task-performance evidence plus documented unguarded-access risk, mixed quality/usability signals, and missing university-level learning evidence → bounded, guardrailed pilot with evaluation, not full adoption.",
82
+ "exceeds_evidence_boundary": [
83
+ "Claiming 'AI coding assistants improve learning' exceeds the boundary: direct learning-effect evidence is missing.",
84
+ "Claiming 'AI works for everyone' exceeds the boundary: population and subject mismatch."
85
+ ]
86
+ }
@@ -0,0 +1,230 @@
1
+ {
2
+ "generated_by": "build_report.py",
3
+ "source": "result.json + result.zh.json",
4
+ "question": "我们准备在大学一年级 C 语言课程中允许学生使用生成式 AI 编程助手。它到底会不会提高学习效果?应该怎样引入?",
5
+ "theme_selected": "claude",
6
+ "theme_display": "Claude Research [Light]",
7
+ "theme_available": [
8
+ "claude",
9
+ "academic",
10
+ "datalab",
11
+ "datalab-dark",
12
+ "presentation"
13
+ ],
14
+ "theme_selection": "generation_time",
15
+ "lang_default": "zh",
16
+ "lang_switchable": [
17
+ "zh",
18
+ "en"
19
+ ],
20
+ "report_pages": [
21
+ "visual_brief",
22
+ "full_report"
23
+ ],
24
+ "full_report_outline": {
25
+ "chapter_count": 6,
26
+ "source": "safe_fallback",
27
+ "chapters": [
28
+ {
29
+ "key": "decision",
30
+ "title_zh": "结论、裁决与研究边界",
31
+ "title_en": "Decision, Adjudication & Research Boundary",
32
+ "modules": [
33
+ "decision",
34
+ "scope"
35
+ ]
36
+ },
37
+ {
38
+ "key": "evidence",
39
+ "title_zh": "关键证据与结果分离",
40
+ "title_en": "Key Evidence & Outcome Separation",
41
+ "modules": [
42
+ "retrieval",
43
+ "outcomes",
44
+ "evidence"
45
+ ]
46
+ },
47
+ {
48
+ "key": "quality",
49
+ "title_zh": "证据可信度、反证与方法审计",
50
+ "title_en": "Evidence Quality, Counterevidence & Method Audit",
51
+ "modules": [
52
+ "quality",
53
+ "conflicts",
54
+ "trace"
55
+ ]
56
+ },
57
+ {
58
+ "key": "action",
59
+ "title_zh": "适用范围与教学行动",
60
+ "title_en": "Applicability & Teaching Action",
61
+ "modules": [
62
+ "applicability",
63
+ "intervention"
64
+ ]
65
+ },
66
+ {
67
+ "key": "evaluation",
68
+ "title_zh": "试点设计、评估与停止条件",
69
+ "title_en": "Pilot, Evaluation & Stop Conditions",
70
+ "modules": [
71
+ "evaluation"
72
+ ]
73
+ },
74
+ {
75
+ "key": "sources",
76
+ "title_zh": "来源、溯源与附录",
77
+ "title_en": "Sources, Traceability & Appendix",
78
+ "modules": [
79
+ "sources"
80
+ ]
81
+ }
82
+ ]
83
+ },
84
+ "visualization_decisions": {
85
+ "outcome_separation": {
86
+ "render": true,
87
+ "reason": "multiple outcome constructs present"
88
+ },
89
+ "outcome_evidence_balance": {
90
+ "render": true,
91
+ "reason": "sufficient effect-direction density"
92
+ },
93
+ "claim_trace": {
94
+ "render": true,
95
+ "reason": "claim-evidence-source relationships present"
96
+ },
97
+ "benchmark": {
98
+ "render": false,
99
+ "reason": "suppressed: absent, simulated, or fewer than two baselines"
100
+ }
101
+ },
102
+ "lieflat_gallery": {
103
+ "layout_source": "deterministic_fallback",
104
+ "selected": [
105
+ {
106
+ "chart_id": "lieflat-bubble-almanac.svg",
107
+ "type": "bubble_almanac",
108
+ "catalog_ref": "L9 Bubble Almanac",
109
+ "source": "evidence.year_x_dimension"
110
+ },
111
+ {
112
+ "chart_id": "lieflat-matrix-heat.svg",
113
+ "type": "matrix_heat",
114
+ "catalog_ref": "L16 Matrix Heat",
115
+ "source": "evidence.year_x_outcome_counts"
116
+ },
117
+ {
118
+ "chart_id": "lieflat-tick-rows.svg",
119
+ "type": "tick_rows",
120
+ "catalog_ref": "F5 Tick Rows",
121
+ "source": "outcomes.direction_counts"
122
+ },
123
+ {
124
+ "chart_id": "lieflat-paired-rungs.svg",
125
+ "type": "paired_rungs",
126
+ "catalog_ref": "F6 Paired Rungs",
127
+ "source": "outcomes.paired_counts"
128
+ },
129
+ {
130
+ "chart_id": "lieflat-brand-spectrum.svg",
131
+ "type": "brand_spectrum",
132
+ "catalog_ref": "L7 Brand Spectrum",
133
+ "source": "outcomes.bipolar_axes"
134
+ },
135
+ {
136
+ "chart_id": "lieflat-hundred-field.svg",
137
+ "type": "hundred_field",
138
+ "catalog_ref": "L14 Hundred Field",
139
+ "source": "evidence.study_type_composition"
140
+ }
141
+ ],
142
+ "suppressed": [],
143
+ "rejected": [],
144
+ "warnings": [
145
+ "visual_layout missing or fully invalid — data-driven fallback selected bubble_almanac, matrix_heat, tick_rows, paired_rungs, brand_spectrum, hundred_field"
146
+ ]
147
+ },
148
+ "charts": [
149
+ {
150
+ "chart_id": "outcome-evidence-overview",
151
+ "purpose": "interactive_analysis",
152
+ "engine": "echarts",
153
+ "data_ref": null,
154
+ "title": "Outcome Evidence Overview",
155
+ "integrity": {
156
+ "numbers_match_result": "NOT_CHECKED",
157
+ "no_axis_distortion": "NOT_CHECKED",
158
+ "no_false_precision": "NOT_CHECKED",
159
+ "colorblind_safe": "NOT_CHECKED"
160
+ }
161
+ },
162
+ {
163
+ "chart_id": "claim-evidence-trace",
164
+ "purpose": "interactive_analysis",
165
+ "engine": "echarts",
166
+ "data_ref": null,
167
+ "title": "Claim-Evidence Trace",
168
+ "integrity": {
169
+ "numbers_match_result": "NOT_CHECKED",
170
+ "no_axis_distortion": "NOT_CHECKED",
171
+ "no_false_precision": "NOT_CHECKED",
172
+ "colorblind_safe": "NOT_CHECKED"
173
+ }
174
+ }
175
+ ],
176
+ "infographics": [
177
+ {
178
+ "chart_id": "workflow",
179
+ "purpose": "process_or_story",
180
+ "engine": "antv_infographic",
181
+ "title": ""
182
+ },
183
+ {
184
+ "chart_id": "tribunal",
185
+ "purpose": "process_or_story",
186
+ "engine": "antv_infographic",
187
+ "title": ""
188
+ },
189
+ {
190
+ "chart_id": "intervention",
191
+ "purpose": "process_or_story",
192
+ "engine": "antv_infographic",
193
+ "title": ""
194
+ },
195
+ {
196
+ "chart_id": "evaluation",
197
+ "purpose": "process_or_story",
198
+ "engine": "antv_infographic",
199
+ "title": ""
200
+ }
201
+ ],
202
+ "academic_figures": [
203
+ {
204
+ "chart_id": "outcome-comparison",
205
+ "purpose": "statistical_publication",
206
+ "engine": "academic_figure",
207
+ "caption": "Fig. 1. Counts of positive / negative / null effects per outcome type (based on effect_direction; publication figure, theme-independent). Source: EduEvidence result.json."
208
+ }
209
+ ],
210
+ "integrity_gate": {
211
+ "status": "PASS",
212
+ "contract_valid": "PASS",
213
+ "claims_bound": "PASS",
214
+ "evidence_bound": 12,
215
+ "sources_resolved": 8,
216
+ "numbers_match_result": "PASS",
217
+ "bilingual_structure_match": "PASS",
218
+ "language_match": "PASS",
219
+ "no_false_precision": "PASS",
220
+ "lieflat_data_bound": "PASS",
221
+ "no_axis_distortion": "NOT_CHECKED",
222
+ "colorblind_safe": "NOT_CHECKED",
223
+ "langs": [
224
+ "zh",
225
+ "en"
226
+ ],
227
+ "generated_by": "build_report.py",
228
+ "source": "result.json + result.zh.json"
229
+ }
230
+ }