eduevidence 5.2.0 → 6.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (276) hide show
  1. package/README.md +54 -42
  2. package/README.zh-CN.md +51 -28
  3. package/SKILL.md +390 -133
  4. package/agents/openai.yaml +4 -0
  5. package/assets/readme/controlled-execution.svg +34 -0
  6. package/assets/readme/logo.png +0 -0
  7. package/assets/readme/research-workflow.svg +56 -0
  8. package/assets/readme/studio-graph.png +0 -0
  9. package/assets/readme/studio-overview.png +0 -0
  10. package/assets/readme/studio-reports.png +0 -0
  11. package/autoevolve/config.yaml +17 -0
  12. package/autoevolve/program.md +25 -0
  13. package/autoevolve/protected.manifest.yaml +34 -0
  14. package/benchmarks/adversarial/cases.jsonl +7 -0
  15. package/benchmarks/evidence-library.json +5268 -0
  16. package/benchmarks/partitions.json +8 -0
  17. package/docs/architecture.md +220 -0
  18. package/docs/autoresearch-evolution-plan.md +2903 -0
  19. package/docs/autoresearch-implementation-status.md +101 -0
  20. package/docs/demo-storyboard.md +20 -0
  21. package/docs/demo-workplace-ai.md +92 -0
  22. package/docs/demo.md +32 -0
  23. package/docs/install-guide.md +150 -0
  24. package/docs/orchestration-role-model.md +1254 -0
  25. package/docs/release-closeout/README.md +17 -0
  26. package/docs/release-closeout/frontend-acceptance.md +23 -0
  27. package/docs/release-closeout/issues.md +19 -0
  28. package/docs/release-closeout/verification.md +28 -0
  29. package/docs/release-contract.md +108 -0
  30. package/docs/research-studio-guide.zh-CN.md +166 -0
  31. package/eduevidence_cli.py +17 -11
  32. package/engine/_resources.py +13 -0
  33. package/engine/autoevolve/__init__.py +3 -0
  34. package/engine/autoevolve/agent_view.py +167 -0
  35. package/engine/autoevolve/core.py +357 -0
  36. package/engine/autoevolve/events.py +11 -0
  37. package/engine/autoevolve/git_workspace.py +77 -0
  38. package/engine/autoevolve/projection.py +23 -0
  39. package/engine/autoevolve/runner.py +413 -0
  40. package/engine/autoevolve/trust.py +146 -0
  41. package/engine/autoresearch/__init__.py +6 -0
  42. package/engine/autoresearch/commit.py +132 -0
  43. package/engine/autoresearch/contracts.py +126 -0
  44. package/engine/autoresearch/controller.py +207 -0
  45. package/engine/autoresearch/events.py +12 -0
  46. package/engine/autoresearch/gap_priority.py +168 -0
  47. package/engine/autoresearch/projection.py +30 -0
  48. package/engine/autoresearch/research_memory.py +59 -0
  49. package/engine/autoresearch/saturation.py +91 -0
  50. package/engine/briefs.py +2 -1
  51. package/engine/capabilities.py +1 -0
  52. package/engine/contracts.py +3 -1
  53. package/engine/evidencecore.py +7 -5
  54. package/engine/gaps.py +90 -51
  55. package/engine/judge_pack.py +65 -0
  56. package/engine/library_builtin.py +3 -1
  57. package/engine/living.py +2 -1
  58. package/engine/meta_synthesis.py +3 -1
  59. package/engine/orchestration.py +460 -0
  60. package/engine/pilot.py +2 -1
  61. package/engine/project.py +2 -2
  62. package/engine/research_service.py +113 -0
  63. package/engine/studio_read_model.py +400 -0
  64. package/engine/tribunal.py +1 -2
  65. package/engine/update.py +1 -0
  66. package/engine/versions.py +1 -1
  67. package/engine/worker_result.py +109 -0
  68. package/engine/workflows.py +70 -0
  69. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1720 -0
  70. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
  71. package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
  72. package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
  73. package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
  74. package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
  75. package/examples/ai-coding-assistant-evidence/frame.json +48 -0
  76. package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
  77. package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
  78. package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
  79. package/examples/ai-coding-assistant-evidence/report_spec.json +219 -0
  80. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2614 -0
  81. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2614 -0
  82. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2614 -0
  83. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2614 -0
  84. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2614 -0
  85. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1720 -0
  86. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1720 -0
  87. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1720 -0
  88. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1720 -0
  89. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1720 -0
  90. package/examples/ai-coding-assistant-evidence/result.json +1453 -0
  91. package/examples/ai-coding-assistant-evidence/result.zh.json +1453 -0
  92. package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
  93. package/examples/ai-coding-assistant-evidence/verdict.json +103 -0
  94. package/examples/workplace-ai-assistant/claims.jsonl +4 -0
  95. package/examples/workplace-ai-assistant/evaluation.json +19 -0
  96. package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
  97. package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
  98. package/examples/workplace-ai-assistant/frame.json +41 -0
  99. package/examples/workplace-ai-assistant/intervention.json +27 -0
  100. package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
  101. package/examples/workplace-ai-assistant/methodology.json +60 -0
  102. package/examples/workplace-ai-assistant/report_spec.json +55 -0
  103. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2484 -0
  104. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2484 -0
  105. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2484 -0
  106. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2484 -0
  107. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2484 -0
  108. package/examples/workplace-ai-assistant/result.json +553 -0
  109. package/examples/workplace-ai-assistant/result.zh.json +553 -0
  110. package/examples/workplace-ai-assistant/search_log.json +19 -0
  111. package/examples/workplace-ai-assistant/sources.jsonl +3 -0
  112. package/examples/workplace-ai-assistant/validation_result.json +9 -0
  113. package/examples/workplace-ai-assistant/verdict.json +52 -0
  114. package/install.sh +7 -7
  115. package/integrations/orchestration_dispatch.py +146 -0
  116. package/package.json +37 -3
  117. package/pyproject.toml +11 -20
  118. package/references/autoresearch.md +30 -0
  119. package/references/evaluation-policy.md +24 -0
  120. package/references/orchestration.md +22 -0
  121. package/references/scientific-invariants.md +19 -0
  122. package/retrieval/audit.py +154 -0
  123. package/schemas/intervention.schema.json +106 -21
  124. package/schemas/report-result.schema.json +9 -1
  125. package/schemas/v2/project.schema.json +2 -2
  126. package/schemas/v2/run.schema.json +1 -1
  127. package/schemas/vNext/autoevolve-session.schema.json +1 -0
  128. package/schemas/vNext/eval-snapshot.schema.json +1 -0
  129. package/schemas/vNext/execution-plan.schema.json +1 -0
  130. package/schemas/vNext/gap-priority.schema.json +1 -0
  131. package/schemas/vNext/negative-search-record.schema.json +1 -0
  132. package/schemas/vNext/research-iteration.schema.json +1 -0
  133. package/schemas/vNext/research-strategy.schema.json +1 -0
  134. package/schemas/vNext/skill-experiment.schema.json +1 -0
  135. package/schemas/vNext/task-spec.schema.json +1 -0
  136. package/schemas/vNext/worker-result.schema.json +1 -0
  137. package/scripts/benchmark_judge.py +2 -2
  138. package/scripts/benchmark_v3.py +26 -43
  139. package/scripts/build_esl_artifacts.py +2 -2
  140. package/scripts/build_evidence_library.py +2 -2
  141. package/scripts/build_gh_pages.py +98 -0
  142. package/scripts/build_readme_diagrams.py +72 -0
  143. package/scripts/build_report_variants.py +85 -0
  144. package/scripts/check_autoresearch_invariants.py +95 -0
  145. package/scripts/daily_evolve.py +30 -0
  146. package/scripts/dashboard_server.py +130 -101
  147. package/scripts/did_regression.py +5 -30
  148. package/scripts/enrich_projects_human_and_lieflat.py +1 -1
  149. package/scripts/generate_metrics.py +4 -3
  150. package/scripts/generate_new_projects.py +1 -1
  151. package/scripts/orchestrator.py +172 -18
  152. package/scripts/rebake_all_5themes.py +1 -2
  153. package/scripts/research_auto_cli.py +475 -0
  154. package/scripts/run_workspace.py +17 -7
  155. package/scripts/search_provenance.py +64 -0
  156. package/scripts/serve_web.py +9 -10
  157. package/scripts/skill_lint.py +1 -1
  158. package/scripts/skill_payload.py +78 -0
  159. package/scripts/validate_schema.py +15 -1
  160. package/scripts/vnext_cli.py +133 -0
  161. package/setup.py +12 -0
  162. package/skill/roles/registry.yaml +45 -0
  163. package/skill/sub-skills/report-generation/SKILL.md +12 -6
  164. package/skill/task-briefs/applicability.md +3 -0
  165. package/skill/task-briefs/projection.md +3 -0
  166. package/skill/workflows/decision-and-pilot.md +10 -0
  167. package/skill/workflows/evaluate-and-update.md +10 -0
  168. package/skill/workflows/evidence-review.md +13 -0
  169. package/visualization/eduevidence-report/assets/base.css +2 -2
  170. package/visualization/eduevidence-report/assets/reader.css +752 -0
  171. package/visualization/eduevidence-report/assets/reader.js +132 -0
  172. package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
  173. package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
  174. package/visualization/eduevidence-report/scripts/build_report.py +58 -65
  175. package/visualization/eduevidence-report/scripts/lieflat_engine.py +24 -100
  176. package/visualization/eduevidence-report/themes/academic.css +1 -1
  177. package/visualization/eduevidence-report/themes/claude.css +1 -1
  178. package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
  179. package/visualization/eduevidence-report/themes/datalab.css +2 -2
  180. package/visualization/eduevidence-report/themes/presentation.css +2 -2
  181. package/web/README.md +18 -0
  182. package/web/index.html +53 -0
  183. package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
  184. package/web/studio/assets/index-CzXocaGv.css +1 -0
  185. package/web/studio/assets/index-pa7jD7n4.js +230 -0
  186. package/web/studio/config.json +1 -0
  187. package/web/studio/index.html +14 -0
  188. package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
  189. package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
  190. package/engine/__pycache__/bias.cpython-312.pyc +0 -0
  191. package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
  192. package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
  193. package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
  194. package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
  195. package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
  196. package/engine/__pycache__/events.cpython-312.pyc +0 -0
  197. package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
  198. package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
  199. package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
  200. package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
  201. package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
  202. package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
  203. package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
  204. package/engine/__pycache__/ids.cpython-312.pyc +0 -0
  205. package/engine/__pycache__/library.cpython-312.pyc +0 -0
  206. package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
  207. package/engine/__pycache__/living.cpython-312.pyc +0 -0
  208. package/engine/__pycache__/log.cpython-312.pyc +0 -0
  209. package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
  210. package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
  211. package/engine/__pycache__/migration.cpython-312.pyc +0 -0
  212. package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
  213. package/engine/__pycache__/paths.cpython-312.pyc +0 -0
  214. package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
  215. package/engine/__pycache__/planner.cpython-312.pyc +0 -0
  216. package/engine/__pycache__/project.cpython-312.pyc +0 -0
  217. package/engine/__pycache__/projections.cpython-312.pyc +0 -0
  218. package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
  219. package/engine/__pycache__/run.cpython-312.pyc +0 -0
  220. package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
  221. package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
  222. package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
  223. package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
  224. package/engine/__pycache__/update.cpython-312.pyc +0 -0
  225. package/engine/__pycache__/versions.cpython-312.pyc +0 -0
  226. package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
  227. package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
  228. package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
  229. package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
  230. package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
  231. package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
  232. package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
  233. package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
  234. package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
  235. package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
  236. package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
  237. package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
  238. package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
  239. package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
  240. package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
  241. package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
  242. package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
  243. package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
  244. package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
  245. package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
  246. package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
  247. package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
  248. package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
  249. package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
  250. package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
  251. package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
  252. package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
  253. package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
  254. package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
  255. package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
  256. package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
  257. package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
  258. package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
  259. package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
  260. package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
  261. package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
  262. package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
  263. package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
  264. package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
  265. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
  266. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
  267. package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
  268. package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
  269. package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
  270. package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
  271. package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
  272. package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
  273. package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
  274. package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
  275. package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
  276. package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
@@ -0,0 +1,1453 @@
1
+ {
2
+ "meta": {
3
+ "skill": "eduevidence",
4
+ "version": "5.2.0",
5
+ "generated_at": "2026-08-24T04:39:14.576951+00:00",
6
+ "mode": "platform_native",
7
+ "question": "我们准备在大学一年级 C 语言课程中允许学生使用生成式 AI 编程助手。它到底会不会提高学习效果?应该怎样引入?",
8
+ "data_origin": "manual_curated"
9
+ },
10
+ "execution": {
11
+ "complexity": "M",
12
+ "mode": "platform_native",
13
+ "agents": [],
14
+ "usage": {
15
+ "measurement_status": "NOT_CAPTURED",
16
+ "input_tokens": null,
17
+ "output_tokens": null,
18
+ "cost_usd": null,
19
+ "latency_s": null
20
+ }
21
+ },
22
+ "research_frame": {
23
+ "question": "我们准备在大学一年级 C 语言课程中允许学生使用生成式 AI 编程助手。它到底会不会提高学习效果?应该怎样引入?",
24
+ "decision_target": "teaching_decision",
25
+ "learner": {
26
+ "education_level": "undergraduate_year_1",
27
+ "major": "computer_science",
28
+ "prior_knowledge": "first_programming_course_no_prior_text_based_programming",
29
+ "special_characteristics": "mixed_ability_large_class_60_students"
30
+ },
31
+ "course": {
32
+ "subject": "C_programming",
33
+ "course_type": "lecture_lab",
34
+ "duration": "16_weeks_one_semester"
35
+ },
36
+ "intervention": {
37
+ "teaching_method": "lecture_with_lab_exercises",
38
+ "ai_tool": "generative_ai_coding_assistant",
39
+ "allowed_usage": "under_design_pending_evidence_review",
40
+ "frequency": "weekly_lab_sessions",
41
+ "duration": "one_semester"
42
+ },
43
+ "comparison": "no_ai_coding_assistant_control",
44
+ "outcomes": {
45
+ "primary": [
46
+ "independent_problem_solving",
47
+ "code_quality"
48
+ ],
49
+ "secondary": [
50
+ "completion_time",
51
+ "retention",
52
+ "knowledge_gain"
53
+ ],
54
+ "risk": [
55
+ "ai_dependency",
56
+ "over_reliance",
57
+ "reduced_transfer"
58
+ ]
59
+ },
60
+ "context": {
61
+ "teacher_support": "TA_supported_two_TAs",
62
+ "class_size": "60_students",
63
+ "online_or_offline": "offline"
64
+ },
65
+ "scope": {
66
+ "time_range": "2021-2026",
67
+ "geography": "worldwide",
68
+ "study_types": [
69
+ "rct",
70
+ "quasi_experimental",
71
+ "observational"
72
+ ]
73
+ },
74
+ "inclusion_criteria": [
75
+ "studies_of_generative_AI_coding_tools_in_learning_to_program",
76
+ "outcomes_measuring_learning_not_only_task_speed",
77
+ "university_or_novice_programming_populations"
78
+ ],
79
+ "exclusion_criteria": [
80
+ "practitioner_anecdotes_without_data",
81
+ "industry_professional_populations_only"
82
+ ],
83
+ "success_condition": "independent problem solving and code quality improve (or do not decline) while AI dependency risk stays controlled; evidence base supports a bounded pilot."
84
+ },
85
+ "decision": {
86
+ "decision_question": "大一 C 语言课程是否应该允许学生使用生成式 AI 编程助手?",
87
+ "target_population": "首次学习 C 语言编程的大一计算机专业学生",
88
+ "target_context": "16 周讲授课+实验课,60 人班级,助教支持,线下",
89
+ "supported_claims": [
90
+ "AI 编程助手在训练期间可靠地提升任务表现(完成速度、正确性)—— E-001、E-006。",
91
+ "无护栏的生成式 AI 访问在移除工具后可能损害独立问题解决能力 —— E-004。",
92
+ "护栏设计(给提示而非给答案)能大幅缓解负面学习效应 —— E-005。",
93
+ "任务表现提升并不自动等于学习提升 —— E-004 与 E-006 的研究内对照。",
94
+ "工具能力可观:Codex 能解出约半数至四分之三的 CS1 考试风格题目 —— E-010。",
95
+ "职业开发者 RCT 显示 Copilot 带来约 55% 任务提速;但职业人群限制直接性 —— E-008。",
96
+ "LLM 代码讲解的质量评级与学生自撰讲解相当,可作支架材料 —— E-011。"
97
+ ],
98
+ "uncertain_claims": [
99
+ "AI 编程助手能否真正改善或保持大学新手的编程学习——本证据集中没有大学层面的直接 RCT [无直接证据]",
100
+ "Kazemitabaar 2023 的一周中性保持性能否延伸到一个学期 —— E-003。",
101
+ "基准质量结论(E-009)与讲解质量评级(E-011)能否转化为课堂学习收益。",
102
+ "可用性研究所记录的理解/所有权困难(E-012)在整学期护栏条件下会如何演变。"
103
+ ],
104
+ "contradicted_claims": [
105
+ "'AI 工具总能提高学习'被 E-004 反驳(无护栏访问,独立考试 −17%)。",
106
+ "'速度收益等于学习收益'被 E-001/E-006/E-008 与 E-004 之间的任务-学习分离所反驳。"
107
+ ],
108
+ "reason_for_disagreement": "分歧来自结果分离(任务 vs 学习)、工具设计(有护栏 vs 无护栏)与人群(K-12/职业者 vs 大学生)。随机实验与基准研究中任务表现证据一致为正;唯一测量移除 AI 后独立表现的研究显示无护栏时有害;可用性与工件研究补充依赖与质量警示而非解决学习问题。",
109
+ "methodology_summary": "八个真实来源:三项随机实验(Kazemitabaar 2023,n=69,K-12;Bastani 2025,n≈950,高中数学;Peng 2023,n=95,职业开发者,预印本)、一项 ESL 写作混合方法研究(Marzuki 2024),以及基准/能力/可用性研究(Yetistiren 2023;Finnie-Ansley 2022;讲解对比 2023;Vaithilingam 2022)。无大学编程课程的直接 RCT。核心 RCT 内部效度强;对大一 C 语言情境的直接性弱。所有来源均带注册表核验 DOI(见 benchmarks/doi-audit/report.md)。",
110
+ "outcome_specific_findings": {
111
+ "completion_time": "训练期与职业任务中为正(E-001、E-008)",
112
+ "independent_problem_solving": "无护栏时中性偏负(E-002、E-004)",
113
+ "retention": "一周内中性(E-003)",
114
+ "assignment_score": "练习期为正、闭卷考试为负(E-004、E-006);工具本身在 CS1 题目可达通过水平(E-010)",
115
+ "code_quality": "基准上结论不一;记录到安全隐患(E-009)",
116
+ "metacognition": "LLM 讲解对比占优(E-011),而新手理解/所有权困难仍存(E-012)",
117
+ "ai_dependency": "无护栏工具下记录到拐杖行为(E-004、E-005、E-012)"
118
+ },
119
+ "short_term_effect": "任务表现可靠提升;无护栏时学习效应中性偏负。",
120
+ "long_term_effect": "无超过一周的证据;长期学习效应未知。",
121
+ "transfer_effect": "无完整迁移证据;一项小样本研究中手动代码修改未受损(E-002)。",
122
+ "risk_effect": "无护栏使用的 AI 依赖与过度依赖风险真实且有记录(E-004),可用性发现亦予印证(E-012)。",
123
+ "applicability": {
124
+ "suitable_for": "在大一 C 课程以护栏化使用政策开展试点",
125
+ "not_suitable_for": "无使用政策的全面放开采用",
126
+ "required_conditions": [
127
+ "护栏化 AI 使用政策(给提示不给答案,仿 GPT Tutor 组)",
128
+ "无 AI 迁移评估",
129
+ "助教支持"
130
+ ]
131
+ },
132
+ "confidence": "Moderate",
133
+ "confidence_breakdown": {
134
+ "score": 0.586,
135
+ "evidence_quality": 0.758,
136
+ "consistency": 0.667,
137
+ "directness": 0.458,
138
+ "evidence_count": 12,
139
+ "independent_studies": 8,
140
+ "independent_samples": 8,
141
+ "count_term": 1.0,
142
+ "conflict_penalty": 0.15,
143
+ "unsupported_penalty": 0.0
144
+ },
145
+ "what_can_be_claimed": [
146
+ "AI 编程助手在训练期提升新手任务表现。",
147
+ "无护栏访问存在损害独立问题解决的真实风险。",
148
+ "护栏设计可以缓解该风险。",
149
+ "大学 C 语言学习的直接证据缺失。",
150
+ "工具能力余量大(CS1 通过率;职业提速 RCT)。"
151
+ ],
152
+ "what_cannot_be_claimed": [
153
+ "AI 编程助手能改善(乃至保持)大学生的编程学习。",
154
+ "任何长期或保持收益。",
155
+ "基于大学样本的任何『哪些学生受益』结论。",
156
+ "基准或可用性发现可替代课堂学习结果。"
157
+ ],
158
+ "missing_evidence": [
159
+ "在大学编程课程中带保持与无 AI 迁移测试的 RCT。",
160
+ "同一课程内变化 AI 使用政策的研究。",
161
+ "跨越一门课的 AI 依赖纵向数据。",
162
+ "职业提速 RCT 的同行评审重复(Peng 等仍为预印本)。"
163
+ ],
164
+ "recommended_action": "pilot",
165
+ "decision_rationale": "任务表现的正面证据 + 有据可查的无护栏风险 + 混合的质量/可用性信号 + 大学层面学习证据缺失 → 有界、护栏化、带评估的试点,而非全面采用。",
166
+ "exceeds_evidence_boundary": [
167
+ "『AI 编程助手提高学习效果』—— 超出边界:缺少直接学习效应证据。",
168
+ "『AI 对所有人都有效』—— 超出边界:人群与学科错配。"
169
+ ],
170
+ "confidence_score": 0.586,
171
+ "confidence_policy_version": "2026-08-12.v2",
172
+ "independent_studies": 8,
173
+ "independent_samples": 8,
174
+ "raw_model_confidence": "Moderate",
175
+ "raw_model_confidence_breakdown": {
176
+ "score": 0.5,
177
+ "evidence_quality": 0.7,
178
+ "consistency": 0.6,
179
+ "directness": 0.4,
180
+ "evidence_count": 12,
181
+ "independent_studies": 8,
182
+ "independent_samples": 8,
183
+ "count_term": 1.0,
184
+ "conflict_penalty": 0.0,
185
+ "unsupported_penalty": 0.0
186
+ }
187
+ },
188
+ "outcomes": [
189
+ {
190
+ "outcome_type": "knowledge_gain",
191
+ "positive_count": 1,
192
+ "negative_count": 0,
193
+ "null_count": 0,
194
+ "evidence_ids": [
195
+ "E-007"
196
+ ]
197
+ },
198
+ {
199
+ "outcome_type": "retention",
200
+ "positive_count": 0,
201
+ "negative_count": 0,
202
+ "null_count": 1,
203
+ "evidence_ids": [
204
+ "E-003"
205
+ ]
206
+ },
207
+ {
208
+ "outcome_type": "independent_problem_solving",
209
+ "positive_count": 0,
210
+ "negative_count": 1,
211
+ "null_count": 2,
212
+ "evidence_ids": [
213
+ "E-002",
214
+ "E-004",
215
+ "E-005"
216
+ ]
217
+ },
218
+ {
219
+ "outcome_type": "completion_time",
220
+ "positive_count": 2,
221
+ "negative_count": 0,
222
+ "null_count": 0,
223
+ "evidence_ids": [
224
+ "E-001",
225
+ "E-008"
226
+ ]
227
+ },
228
+ {
229
+ "outcome_type": "code_quality",
230
+ "positive_count": 0,
231
+ "negative_count": 0,
232
+ "null_count": 1,
233
+ "evidence_ids": [
234
+ "E-009"
235
+ ]
236
+ },
237
+ {
238
+ "outcome_type": "assignment_score",
239
+ "positive_count": 2,
240
+ "negative_count": 0,
241
+ "null_count": 0,
242
+ "evidence_ids": [
243
+ "E-006",
244
+ "E-010"
245
+ ]
246
+ },
247
+ {
248
+ "outcome_type": "metacognition",
249
+ "positive_count": 1,
250
+ "negative_count": 0,
251
+ "null_count": 0,
252
+ "evidence_ids": [
253
+ "E-011"
254
+ ]
255
+ },
256
+ {
257
+ "outcome_type": "over_reliance",
258
+ "positive_count": 0,
259
+ "negative_count": 1,
260
+ "null_count": 0,
261
+ "evidence_ids": [
262
+ "E-012"
263
+ ]
264
+ }
265
+ ],
266
+ "outcome_mapping": {
267
+ "entries": [
268
+ {
269
+ "outcome_type": "ai_dependency",
270
+ "declared_in_frame": true,
271
+ "status": "no_evidence",
272
+ "support_count": 0,
273
+ "contradict_count": 0,
274
+ "neutral_count": 0,
275
+ "evidence_ids": []
276
+ },
277
+ {
278
+ "outcome_type": "assignment_score",
279
+ "declared_in_frame": false,
280
+ "status": "supported",
281
+ "support_count": 2,
282
+ "contradict_count": 0,
283
+ "neutral_count": 0,
284
+ "evidence_ids": [
285
+ "E-006",
286
+ "E-010"
287
+ ]
288
+ },
289
+ {
290
+ "outcome_type": "code_quality",
291
+ "declared_in_frame": true,
292
+ "status": "null_evidence_only",
293
+ "support_count": 0,
294
+ "contradict_count": 0,
295
+ "neutral_count": 1,
296
+ "evidence_ids": [
297
+ "E-009"
298
+ ]
299
+ },
300
+ {
301
+ "outcome_type": "completion_time",
302
+ "declared_in_frame": true,
303
+ "status": "supported",
304
+ "support_count": 2,
305
+ "contradict_count": 0,
306
+ "neutral_count": 0,
307
+ "evidence_ids": [
308
+ "E-001",
309
+ "E-008"
310
+ ]
311
+ },
312
+ {
313
+ "outcome_type": "independent_problem_solving",
314
+ "declared_in_frame": true,
315
+ "status": "contested",
316
+ "support_count": 1,
317
+ "contradict_count": 1,
318
+ "neutral_count": 1,
319
+ "evidence_ids": [
320
+ "E-002",
321
+ "E-004",
322
+ "E-005"
323
+ ]
324
+ },
325
+ {
326
+ "outcome_type": "knowledge_gain",
327
+ "declared_in_frame": true,
328
+ "status": "supported",
329
+ "support_count": 1,
330
+ "contradict_count": 0,
331
+ "neutral_count": 0,
332
+ "evidence_ids": [
333
+ "E-007"
334
+ ]
335
+ },
336
+ {
337
+ "outcome_type": "metacognition",
338
+ "declared_in_frame": false,
339
+ "status": "supported",
340
+ "support_count": 1,
341
+ "contradict_count": 0,
342
+ "neutral_count": 0,
343
+ "evidence_ids": [
344
+ "E-011"
345
+ ]
346
+ },
347
+ {
348
+ "outcome_type": "over_reliance",
349
+ "declared_in_frame": true,
350
+ "status": "contradicted",
351
+ "support_count": 0,
352
+ "contradict_count": 1,
353
+ "neutral_count": 0,
354
+ "evidence_ids": [
355
+ "E-012"
356
+ ]
357
+ },
358
+ {
359
+ "outcome_type": "reduced_transfer",
360
+ "declared_in_frame": true,
361
+ "status": "no_evidence",
362
+ "support_count": 0,
363
+ "contradict_count": 0,
364
+ "neutral_count": 0,
365
+ "evidence_ids": []
366
+ },
367
+ {
368
+ "outcome_type": "retention",
369
+ "declared_in_frame": true,
370
+ "status": "null_evidence_only",
371
+ "support_count": 0,
372
+ "contradict_count": 0,
373
+ "neutral_count": 1,
374
+ "evidence_ids": [
375
+ "E-003"
376
+ ]
377
+ }
378
+ ],
379
+ "declared_without_evidence": [
380
+ "ai_dependency",
381
+ "reduced_transfer"
382
+ ]
383
+ },
384
+ "claims": [
385
+ {
386
+ "claim": "在练习环节,无护栏的 GPT Base(类标准 ChatGPT 界面)使高中生的练习成绩相对对照组提高 48%,带护栏的 GPT Tutor 提高 127%(Table 1:practice 系数 0.137/0.361,对照均值 0.284)",
387
+ "outcome_type": "completion_time",
388
+ "claim_id": "C-001",
389
+ "evidence_ids": [
390
+ "E-001"
391
+ ],
392
+ "status": "SUPPORTED"
393
+ },
394
+ {
395
+ "claim": "移除 AI 访问后的无辅助独立考试中,GPT Base 组成绩比从未使用 AI 的对照组低 17%(统计显著),表明无护栏使用 AI 损害技能习得;机制上学生把 GPT 当'拐杖'直接抄答案(GPT Base 答对率仅 51%,其中 42% 逻辑错误、8% 算术错误),且学生自评过度乐观、未察觉学习受损",
396
+ "outcome_type": "independent_problem_solving",
397
+ "claim_id": "C-002",
398
+ "evidence_ids": [
399
+ "E-002"
400
+ ],
401
+ "status": "SUPPORTED"
402
+ },
403
+ {
404
+ "claim": "带护栏的 GPT Tutor(教师设计提示而非直接答案)在练习成绩 +127% 的同时,移除访问后的独立考试负效应基本消除(-0.004,不显著),说明精心设计的护栏可兼得练习提升与学习保持",
405
+ "outcome_type": "retention",
406
+ "claim_id": "C-003",
407
+ "evidence_ids": [
408
+ "E-003"
409
+ ],
410
+ "status": "SUPPORTED"
411
+ },
412
+ {
413
+ "claim": "训练阶段使用 OpenAI Codex 的 10-17 岁新手在 45 道 Python 代码编写任务上表现显著提升:完成率提高 1.15 倍、得分提高 1.8 倍",
414
+ "outcome_type": "independent_problem_solving",
415
+ "claim_id": "C-004",
416
+ "evidence_ids": [
417
+ "E-004"
418
+ ],
419
+ "status": "SUPPORTED"
420
+ },
421
+ {
422
+ "claim": "训练期使用 Codex 的学习者一周后评估后测成绩略好于对照组,但差异未达统计显著(保持力无显著差异);Scratch 前测高分者若有 Codex 使用史,保持后测显著更好",
423
+ "outcome_type": "independent_problem_solving",
424
+ "claim_id": "C-005",
425
+ "evidence_ids": [
426
+ "E-005"
427
+ ],
428
+ "status": "SUPPORTED"
429
+ },
430
+ {
431
+ "claim": "质性案例研究中,3 名 EFL 学生珍视 ChatGPT 的辅助价值(消除不确定性、澄清词汇、提供内容建议、语法/结构反馈,让学生专注于创意层面),并形成语言精修、观点生成与结构、校对与信心增强等使用策略",
432
+ "outcome_type": "assignment_score",
433
+ "claim_id": "C-006",
434
+ "evidence_ids": [
435
+ "E-006"
436
+ ],
437
+ "status": "SUPPORTED"
438
+ },
439
+ {
440
+ "claim": "同一质性研究中,学生担忧 AI 使用的学术真实性与过度依赖风险(建议过于复杂/正式、语气不符、文化刻板印象等局限),强调必须保持人的判断并寻求教师/同伴反馈,呼吁伦理指引与批判性思维培养",
441
+ "outcome_type": "knowledge_gain",
442
+ "claim_id": "C-007",
443
+ "evidence_ids": [
444
+ "E-007"
445
+ ],
446
+ "status": "SUPPORTED"
447
+ },
448
+ {
449
+ "claim": "随机对照实验(n=95)显示:使用 Copilot 的职业开发者完成标准化编程任务的用时比对照组缩短约 55%。",
450
+ "outcome_type": "completion_time",
451
+ "claim_id": "C-008",
452
+ "evidence_ids": [
453
+ "E-008"
454
+ ],
455
+ "status": "SUPPORTED"
456
+ },
457
+ {
458
+ "claim": "系统性基准评估显示 Copilot 生成代码相对人类代码的质量结论不一:部分正确性具竞争力,同时记录到安全相关缺陷。",
459
+ "outcome_type": "code_quality",
460
+ "claim_id": "C-009",
461
+ "evidence_ids": [
462
+ "E-009"
463
+ ],
464
+ "status": "SUPPORTED"
465
+ },
466
+ {
467
+ "claim": "Codex 在 CS1 考试风格题目上能给出通过水平的解答(依数据集约 50%–75%),表明新手手中存在可观的任务能力余量。",
468
+ "outcome_type": "assignment_score",
469
+ "claim_id": "C-010",
470
+ "evidence_ids": [
471
+ "E-010"
472
+ ],
473
+ "status": "SUPPORTED"
474
+ },
475
+ {
476
+ "claim": "受控比较发现 LLM 生成的代码讲解与学生自撰讲解相当(部分更优),适合作为解释性支架材料,而非替代学生的解释练习。",
477
+ "outcome_type": "metacognition",
478
+ "claim_id": "C-011",
479
+ "evidence_ids": [
480
+ "E-011"
481
+ ],
482
+ "status": "SUPPORTED"
483
+ },
484
+ {
485
+ "claim": "尽管首任务完成更快,参与者难以理解并调试 AI 生成的解法、对最终程序所有权感低——记录了纯速度指标遗漏的认知与依赖风险。",
486
+ "outcome_type": "over_reliance",
487
+ "claim_id": "C-012",
488
+ "evidence_ids": [
489
+ "E-012"
490
+ ],
491
+ "status": "CONTRADICT"
492
+ }
493
+ ],
494
+ "sources": [
495
+ {
496
+ "source_id": "S-2023-kazemitabaar",
497
+ "title": "Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming",
498
+ "year": 2023,
499
+ "doi": "10.1145/3544548.3580919",
500
+ "canonical_url": "https://dl.acm.org/doi/10.1145/3544548.3580919",
501
+ "authority_level": "tier1_paper_doi",
502
+ "source_location": "https://dl.acm.org/doi/10.1145/3544548.3580919",
503
+ "doi_verified": true,
504
+ "retracted": false
505
+ },
506
+ {
507
+ "source_id": "S-2025-bastani",
508
+ "title": "Generative AI without guardrails can harm learning: Evidence from high school mathematics",
509
+ "year": 2025,
510
+ "doi": "10.1073/pnas.2422633122",
511
+ "canonical_url": "https://www.pnas.org/doi/10.1073/pnas.2422633122",
512
+ "authority_level": "tier1_paper_doi",
513
+ "source_location": "https://www.pnas.org/doi/10.1073/pnas.2422633122",
514
+ "doi_verified": true,
515
+ "retracted": false
516
+ },
517
+ {
518
+ "source_id": "S-2024-marzuki",
519
+ "title": "Impact of ChatGPT on ESL students' academic writing skills",
520
+ "year": 2024,
521
+ "doi": "10.1186/s40561-024-00295-9",
522
+ "canonical_url": "https://link.springer.com/article/10.1186/s40561-024-00295-9",
523
+ "authority_level": "tier1_paper_doi",
524
+ "source_location": "https://link.springer.com/article/10.1186/s40561-024-00295-9",
525
+ "doi_verified": true,
526
+ "retracted": false
527
+ },
528
+ {
529
+ "source_id": "S-2023-peng",
530
+ "title": "The Impact of AI on Developer Productivity: Evidence from GitHub Copilot",
531
+ "year": 2023,
532
+ "doi": "10.48550/arXiv.2302.06590",
533
+ "canonical_url": "https://doi.org/10.48550/arXiv.2302.06590",
534
+ "authority_level": "tier2_academic_database",
535
+ "source_location": "https://arxiv.org/abs/2302.06590",
536
+ "doi_verified": true,
537
+ "retracted": false
538
+ },
539
+ {
540
+ "source_id": "S-2023-yetistiren",
541
+ "title": "GitHub Copilot AI Pair Programmer: Asset or Liability?",
542
+ "year": 2023,
543
+ "doi": "10.1016/j.jss.2023.111734",
544
+ "canonical_url": "https://doi.org/10.1016/j.jss.2023.111734",
545
+ "authority_level": "tier1_paper_doi",
546
+ "source_location": "https://doi.org/10.1016/j.jss.2023.111734",
547
+ "doi_verified": true,
548
+ "retracted": false
549
+ },
550
+ {
551
+ "source_id": "S-2022-finnie-ansley",
552
+ "title": "Using GitHub Copilot to Solve Introductory Programming Problems",
553
+ "year": 2022,
554
+ "doi": "10.1145/3545945.3569830",
555
+ "canonical_url": "https://dl.acm.org/doi/10.1145/3545945.3569830",
556
+ "authority_level": "tier1_paper_doi",
557
+ "source_location": "https://dl.acm.org/doi/10.1145/3545945.3569830",
558
+ "doi_verified": true,
559
+ "retracted": false
560
+ },
561
+ {
562
+ "source_id": "S-2023-explanations-compare",
563
+ "title": "Comparing Code Explanations Created by Students and Large Language Models",
564
+ "year": 2023,
565
+ "doi": "10.1145/3587102.3588785",
566
+ "canonical_url": "https://dl.acm.org/doi/10.1145/3587102.3588785",
567
+ "authority_level": "tier1_paper_doi",
568
+ "source_location": "https://dl.acm.org/doi/10.1145/3587102.3588785",
569
+ "doi_verified": true,
570
+ "retracted": false
571
+ },
572
+ {
573
+ "source_id": "S-2022-vaithilingam",
574
+ "title": "Expectation vs. Experience: Evaluating the Usability of Code Generation Tools",
575
+ "year": 2022,
576
+ "doi": "10.1145/3491101.3519665",
577
+ "canonical_url": "https://dl.acm.org/doi/10.1145/3491101.3519665",
578
+ "authority_level": "tier1_paper_doi",
579
+ "source_location": "https://dl.acm.org/doi/10.1145/3491101.3519665",
580
+ "doi_verified": true,
581
+ "retracted": false
582
+ }
583
+ ],
584
+ "evidence": [
585
+ {
586
+ "evidence_id": "E-001",
587
+ "source_id": "S-2023-kazemitabaar",
588
+ "title": "Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming",
589
+ "year": 2023,
590
+ "study_type": "rct",
591
+ "education_level": "k12_ages_10_17",
592
+ "subject": "introductory_python",
593
+ "population": "土耳其高中 9-11 年级数学学生(约 1000 名学生、4 次 90 分钟课内环节,共 2848 观测)",
594
+ "sample_size": 69,
595
+ "intervention": "三臂 RCT:GPT Base(无护栏标准 ChatGPT 式界面)与 GPT Tutor(护栏版,教师设计提示、不给直接答案)用于数学练习",
596
+ "comparison": "对照组(无 AI 传统教学)",
597
+ "outcome_type": "completion_time",
598
+ "outcome_measure": "code_authoring_task_progress_and_time",
599
+ "claim": "在练习环节,无护栏的 GPT Base(类标准 ChatGPT 界面)使高中生的练习成绩相对对照组提高 48%,带护栏的 GPT Tutor 提高 127%(Table 1:practice 系数 0.137/0.361,对照均值 0.284)",
600
+ "direction": "support",
601
+ "relation_to_claim": "support",
602
+ "effect_direction": "positive",
603
+ "study_id": "STUDY-KAZEMITABAAR-2023",
604
+ "sample_id": "SMPL-KAZEMITABAAR-2023-N69",
605
+ "effect": "1.15x completion rate, 0.57x time, 1.8x correctness",
606
+ "duration": "3_weeks_training",
607
+ "method": "controlled experiment with random assignment, immediate post-test and 1-week retention test",
608
+ "strengths": [
609
+ "randomized_controlled_design",
610
+ "immediate_post_test_and_retention_test",
611
+ "code_modification_task_guard"
612
+ ],
613
+ "limitations": [
614
+ "non_university_population_ages_10_17",
615
+ "small_sample_69",
616
+ "self-paced environment differs from classroom"
617
+ ],
618
+ "confounders": [
619
+ "prior_programming_competency_interaction"
620
+ ],
621
+ "source_location": "https://dl.acm.org/doi/10.1145/3544548.3580919",
622
+ "quality_dimensions": {
623
+ "D1_study_design": 2,
624
+ "D2_sample_quality": 2,
625
+ "D3_measurement_validity": 2,
626
+ "D4_temporal_strength": 2,
627
+ "D5_directness": 1
628
+ },
629
+ "quality_score": 9.0,
630
+ "evidence_level": "strong",
631
+ "applicability": {
632
+ "learner_match": "partial_novice_programmers_but_younger",
633
+ "subject_match": "introductory_programming",
634
+ "tool_match": "codex_like_generative_ai",
635
+ "scope": "task_performance_during_training"
636
+ },
637
+ "confidence": 0.7,
638
+ "status": "SUPPORTED",
639
+ "decision_relation": "support_adoption",
640
+ "claim_id": "C-001"
641
+ },
642
+ {
643
+ "evidence_id": "E-002",
644
+ "source_id": "S-2023-kazemitabaar",
645
+ "title": "Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming",
646
+ "year": 2023,
647
+ "study_type": "rct",
648
+ "education_level": "k12_ages_10_17",
649
+ "subject": "introductory_python",
650
+ "population": "土耳其高中 9-11 年级数学学生(约 1000 名学生,共 2848 观测)",
651
+ "sample_size": 69,
652
+ "intervention": "无护栏 GPT Base(类标准 ChatGPT 界面)课内练习;移除访问后参加独立考试",
653
+ "comparison": "对照组(从未使用 AI)",
654
+ "outcome_type": "independent_problem_solving",
655
+ "outcome_measure": "manual code-modification tasks during training",
656
+ "claim": "移除 AI 访问后的无辅助独立考试中,GPT Base 组成绩比从未使用 AI 的对照组低 17%(统计显著),表明无护栏使用 AI 损害技能习得;机制上学生把 GPT 当'拐杖'直接抄答案(GPT Base 答对率仅 51%,其中 42% 逻辑错误、8% 算术错误),且学生自评过度乐观、未察觉学习受损",
657
+ "direction": "neutral",
658
+ "relation_to_claim": "neutral",
659
+ "effect_direction": "null",
660
+ "study_id": "STUDY-KAZEMITABAAR-2023",
661
+ "sample_id": "SMPL-KAZEMITABAAR-2023-N69",
662
+ "effect": "no significant difference between groups",
663
+ "duration": "3_weeks_training",
664
+ "method": "controlled experiment, code-modification task followed each code-authoring task",
665
+ "strengths": [
666
+ "direct_test_of_transfer-adjacent_skill",
667
+ "same_session_measurement"
668
+ ],
669
+ "limitations": [
670
+ "code modification is not full independent problem solving",
671
+ "non_university population"
672
+ ],
673
+ "confounders": [
674
+ "practice_effect"
675
+ ],
676
+ "source_location": "https://dl.acm.org/doi/10.1145/3544548.3580919",
677
+ "quality_dimensions": {
678
+ "D1_study_design": 2,
679
+ "D2_sample_quality": 2,
680
+ "D3_measurement_validity": 1,
681
+ "D4_temporal_strength": 1,
682
+ "D5_directness": 1
683
+ },
684
+ "quality_score": 7.0,
685
+ "evidence_level": "moderate",
686
+ "applicability": {
687
+ "learner_match": "partial",
688
+ "subject_match": "introductory_programming",
689
+ "tool_match": "codex",
690
+ "scope": "short-term manual code modification"
691
+ },
692
+ "confidence": 0.5,
693
+ "status": "SUPPORTED",
694
+ "decision_relation": "neutral",
695
+ "claim_id": "C-002"
696
+ },
697
+ {
698
+ "evidence_id": "E-003",
699
+ "source_id": "S-2023-kazemitabaar",
700
+ "title": "Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming",
701
+ "year": 2023,
702
+ "study_type": "rct",
703
+ "education_level": "k12_ages_10_17",
704
+ "subject": "introductory_python",
705
+ "population": "土耳其高中 9-11 年级数学学生(约 1000 名学生,共 2848 观测)",
706
+ "sample_size": 69,
707
+ "intervention": "GPT Tutor(护栏版:教师设计提示、不给直接答案)用于数学练习",
708
+ "comparison": "对照组(无 AI)与 GPT Base(无护栏)组",
709
+ "outcome_type": "retention",
710
+ "outcome_measure": "retention post-test one week after training",
711
+ "claim": "带护栏的 GPT Tutor(教师设计提示而非直接答案)在练习成绩 +127% 的同时,移除访问后的独立考试负效应基本消除(-0.004,不显著),说明精心设计的护栏可兼得练习提升与学习保持",
712
+ "direction": "neutral",
713
+ "relation_to_claim": "neutral",
714
+ "effect_direction": "null",
715
+ "study_id": "STUDY-KAZEMITABAAR-2023",
716
+ "sample_id": "SMPL-KAZEMITABAAR-2023-N69",
717
+ "effect": "slightly better for Codex group but not significant",
718
+ "duration": "3_weeks_training_plus_1_week_retention",
719
+ "method": "controlled experiment with delayed retention test",
720
+ "strengths": [
721
+ "delayed_test_included"
722
+ ],
723
+ "limitations": [
724
+ "1-week retention window is short",
725
+ "small sample",
726
+ "non-university population"
727
+ ],
728
+ "confounders": [
729
+ "prior_competency"
730
+ ],
731
+ "source_location": "https://dl.acm.org/doi/10.1145/3544548.3580919",
732
+ "quality_dimensions": {
733
+ "D1_study_design": 2,
734
+ "D2_sample_quality": 2,
735
+ "D3_measurement_validity": 2,
736
+ "D4_temporal_strength": 2,
737
+ "D5_directness": 1
738
+ },
739
+ "quality_score": 9.0,
740
+ "evidence_level": "strong",
741
+ "applicability": {
742
+ "learner_match": "partial",
743
+ "subject_match": "introductory_programming",
744
+ "tool_match": "codex",
745
+ "scope": "retention over one week"
746
+ },
747
+ "confidence": 0.5,
748
+ "status": "SUPPORTED",
749
+ "decision_relation": "neutral",
750
+ "claim_id": "C-003"
751
+ },
752
+ {
753
+ "evidence_id": "E-004",
754
+ "source_id": "S-2025-bastani",
755
+ "title": "Generative AI without guardrails can harm learning: Evidence from high school mathematics",
756
+ "year": 2025,
757
+ "study_type": "rct",
758
+ "education_level": "high_school",
759
+ "subject": "mathematics",
760
+ "population": "69 名 10-17 岁编程新手(含高中/初中年龄段)",
761
+ "sample_size": 950,
762
+ "intervention": "训练阶段一半学习者可使用 OpenAI Codex 完成代码编写任务,任务后接代码修改任务",
763
+ "comparison": "无 Codex 访问组",
764
+ "outcome_type": "independent_problem_solving",
765
+ "outcome_measure": "exam without access to AI resources after practice phase",
766
+ "claim": "训练阶段使用 OpenAI Codex 的 10-17 岁新手在 45 道 Python 代码编写任务上表现显著提升:完成率提高 1.15 倍、得分提高 1.8 倍",
767
+ "direction": "contradict",
768
+ "relation_to_claim": "contradict",
769
+ "effect_direction": "negative",
770
+ "study_id": "STUDY-BASTANI-2025",
771
+ "sample_id": "SMPL-BASTANI-2025-N950",
772
+ "effect": "negative_17_percent_on_independent_exam",
773
+ "duration": "in_class_study_sessions",
774
+ "method": "large-scale randomized controlled trial, practice phase then closed-book exam",
775
+ "strengths": [
776
+ "large_scale_rct",
777
+ "independent_exam_without_ai",
778
+ "arm_wise_design"
779
+ ],
780
+ "limitations": [
781
+ "high_school_mathematics_not_university_programming",
782
+ "single_country"
783
+ ],
784
+ "confounders": [
785
+ "tool_design_difference"
786
+ ],
787
+ "source_location": "https://www.pnas.org/doi/10.1073/pnas.2422633122",
788
+ "quality_dimensions": {
789
+ "D1_study_design": 2,
790
+ "D2_sample_quality": 2,
791
+ "D3_measurement_validity": 2,
792
+ "D4_temporal_strength": 1,
793
+ "D5_directness": 1
794
+ },
795
+ "quality_score": 8.0,
796
+ "evidence_level": "strong",
797
+ "applicability": {
798
+ "learner_match": "partial_same_age_band_different_subject",
799
+ "subject_match": "no_mathematics_vs_programming",
800
+ "tool_match": "gpt4_chat_interface",
801
+ "scope": "unguarded_general_chat_interface"
802
+ },
803
+ "confidence": 0.75,
804
+ "status": "SUPPORTED",
805
+ "decision_relation": "oppose_adoption",
806
+ "claim_id": "C-004"
807
+ },
808
+ {
809
+ "evidence_id": "E-005",
810
+ "source_id": "S-2025-bastani",
811
+ "title": "Generative AI without guardrails can harm learning: Evidence from high school mathematics",
812
+ "year": 2025,
813
+ "study_type": "rct",
814
+ "education_level": "high_school",
815
+ "subject": "mathematics",
816
+ "population": "69 名 10-17 岁编程新手",
817
+ "sample_size": 950,
818
+ "intervention": "训练阶段使用 OpenAI Codex 完成代码编写任务",
819
+ "comparison": "无 Codex 访问组;一周后评估后测",
820
+ "outcome_type": "independent_problem_solving",
821
+ "outcome_measure": "exam without access to AI resources after practice phase",
822
+ "claim": "训练期使用 Codex 的学习者一周后评估后测成绩略好于对照组,但差异未达统计显著(保持力无显著差异);Scratch 前测高分者若有 Codex 使用史,保持后测显著更好",
823
+ "direction": "support",
824
+ "relation_to_claim": "support",
825
+ "effect_direction": "null",
826
+ "study_id": "STUDY-BASTANI-2025",
827
+ "sample_id": "SMPL-BASTANI-2025-N950",
828
+ "effect": "negative effect essentially eradicated, no positive effect observed",
829
+ "duration": "in_class_study_sessions",
830
+ "method": "large-scale randomized controlled trial, three arms",
831
+ "strengths": [
832
+ "direct_manipulation_of_tool_design",
833
+ "large_sample"
834
+ ],
835
+ "limitations": [
836
+ "no_positive_learning_gain_even_with_guardrails",
837
+ "subject_mismatch"
838
+ ],
839
+ "confounders": [
840
+ "prompt_engineering_effort"
841
+ ],
842
+ "source_location": "https://www.pnas.org/doi/10.1073/pnas.2422633122",
843
+ "quality_dimensions": {
844
+ "D1_study_design": 2,
845
+ "D2_sample_quality": 2,
846
+ "D3_measurement_validity": 2,
847
+ "D4_temporal_strength": 1,
848
+ "D5_directness": 1
849
+ },
850
+ "quality_score": 8.0,
851
+ "evidence_level": "strong",
852
+ "applicability": {
853
+ "learner_match": "partial",
854
+ "subject_match": "no",
855
+ "tool_match": "guardrailed_tutor_design",
856
+ "scope": "guardrail_design_principle_transferable"
857
+ },
858
+ "confidence": 0.75,
859
+ "status": "SUPPORTED",
860
+ "decision_relation": "conditional",
861
+ "claim_id": "C-005"
862
+ },
863
+ {
864
+ "evidence_id": "E-006",
865
+ "source_id": "S-2025-bastani",
866
+ "title": "Generative AI without guardrails can harm learning: Evidence from high school mathematics",
867
+ "year": 2025,
868
+ "study_type": "rct",
869
+ "education_level": "high_school",
870
+ "subject": "mathematics",
871
+ "population": "3 名不同水平(R1-R3)的 EFL 学生,半结构化访谈",
872
+ "sample_size": 950,
873
+ "intervention": "学生在学术写作过程中使用 ChatGPT 的体验与策略(质性研究,无效应量测量)",
874
+ "comparison": "无对照组(质性案例研究)",
875
+ "outcome_type": "assignment_score",
876
+ "outcome_measure": "practice problem performance during study sessions",
877
+ "claim": "质性案例研究中,3 名 EFL 学生珍视 ChatGPT 的辅助价值(消除不确定性、澄清词汇、提供内容建议、语法/结构反馈,让学生专注于创意层面),并形成语言精修、观点生成与结构、校对与信心增强等使用策略",
878
+ "direction": "support",
879
+ "relation_to_claim": "support",
880
+ "effect_direction": "positive",
881
+ "study_id": "STUDY-BASTANI-2025",
882
+ "sample_id": "SMPL-BASTANI-2025-N950",
883
+ "effect": "48-127 percent improvement on practice problems",
884
+ "duration": "in_class_study_sessions",
885
+ "method": "randomized controlled trial with practice and closed-book exam phases",
886
+ "strengths": [
887
+ "same_study_compares_task_and_learning",
888
+ "large_sample"
889
+ ],
890
+ "limitations": [
891
+ "subject_mismatch_mathematics"
892
+ ],
893
+ "confounders": [
894
+ "task_familiarity"
895
+ ],
896
+ "source_location": "https://www.pnas.org/doi/10.1073/pnas.2422633122",
897
+ "quality_dimensions": {
898
+ "D1_study_design": 2,
899
+ "D2_sample_quality": 2,
900
+ "D3_measurement_validity": 2,
901
+ "D4_temporal_strength": 1,
902
+ "D5_directness": 1
903
+ },
904
+ "quality_score": 8.0,
905
+ "evidence_level": "strong",
906
+ "applicability": {
907
+ "learner_match": "partial",
908
+ "subject_match": "no",
909
+ "tool_match": "gpt4",
910
+ "scope": "task_performance_vs_learning_separation"
911
+ },
912
+ "confidence": 0.75,
913
+ "status": "SUPPORTED",
914
+ "decision_relation": "conditional",
915
+ "claim_id": "C-006"
916
+ },
917
+ {
918
+ "evidence_id": "E-007",
919
+ "source_id": "S-2024-marzuki",
920
+ "title": "Impact of ChatGPT on ESL students' academic writing skills",
921
+ "year": 2024,
922
+ "study_type": "mixed_methods",
923
+ "education_level": "undergraduate",
924
+ "subject": "academic_writing_esl",
925
+ "population": "3 名不同水平的 EFL 学生,半结构化访谈",
926
+ "sample_size": 72,
927
+ "intervention": "学生在学术写作过程中使用 ChatGPT 的体验与策略",
928
+ "comparison": "无对照组(质性案例研究)",
929
+ "outcome_type": "knowledge_gain",
930
+ "outcome_measure": "writing tests with pre-post-delayed design",
931
+ "claim": "同一质性研究中,学生担忧 AI 使用的学术真实性与过度依赖风险(建议过于复杂/正式、语气不符、文化刻板印象等局限),强调必须保持人的判断并寻求教师/同伴反馈,呼吁伦理指引与批判性思维培养",
932
+ "direction": "support",
933
+ "relation_to_claim": "support",
934
+ "effect_direction": "positive",
935
+ "study_id": "STUDY-MARZUKI-2024",
936
+ "sample_id": "SMPL-MARZUKI-2024-N72",
937
+ "effect": "significant positive impact on writing skills",
938
+ "duration": "6_hours_intervention",
939
+ "method": "mixed methods intervention study, pre/post/delayed tests and focus groups",
940
+ "strengths": [
941
+ "delayed_post_test",
942
+ "mixed_methods_triangulation"
943
+ ],
944
+ "limitations": [
945
+ "short_intervention_6_hours",
946
+ "single_institution",
947
+ "elite_private_university"
948
+ ],
949
+ "confounders": [
950
+ "self_selection_consent"
951
+ ],
952
+ "source_location": "https://link.springer.com/article/10.1186/s40561-024-00295-9",
953
+ "quality_dimensions": {
954
+ "D1_study_design": 1,
955
+ "D2_sample_quality": 1,
956
+ "D3_measurement_validity": 2,
957
+ "D4_temporal_strength": 2,
958
+ "D5_directness": 0
959
+ },
960
+ "quality_score": 6.0,
961
+ "evidence_level": "moderate",
962
+ "applicability": {
963
+ "learner_match": "yes_undergraduate",
964
+ "subject_match": "no_writing_not_programming",
965
+ "tool_match": "chatgpt",
966
+ "scope": "formative_feedback_writing"
967
+ },
968
+ "confidence": 0.55,
969
+ "status": "SUPPORTED",
970
+ "decision_relation": "support_adoption",
971
+ "claim_id": "C-007"
972
+ },
973
+ {
974
+ "evidence_id": "E-008",
975
+ "source_id": "S-2023-peng",
976
+ "title": "The Impact of AI on Developer Productivity: Evidence from GitHub Copilot",
977
+ "year": 2023,
978
+ "study_type": "rct",
979
+ "education_level": "professional_developers_not_students",
980
+ "subject": "standardized_javascript_http_server_task",
981
+ "population": "95 名经自由职业平台招募的职业开发者,完成标准化编码任务",
982
+ "sample_size": 95,
983
+ "intervention": "任务期间可使用 GitHub Copilot",
984
+ "comparison": "不可使用 Copilot 的对照组",
985
+ "outcome_type": "completion_time",
986
+ "outcome_measure": "time_to_complete_http_server_implementation",
987
+ "claim": "随机对照实验(n=95)显示:使用 Copilot 的职业开发者完成标准化编程任务的用时比对照组缩短约 55%。",
988
+ "direction": "support",
989
+ "relation_to_claim": "support",
990
+ "effect_direction": "positive",
991
+ "study_id": "STUDY-PENG-2023",
992
+ "sample_id": "SMPL-PENG-2023-N95",
993
+ "effect": "~55.8% faster task completion in Copilot group",
994
+ "duration": "single_task_session",
995
+ "method": "online randomized controlled experiment with objective completion-time metric",
996
+ "strengths": [
997
+ "randomized_controlled_design",
998
+ "objective_completion_time_metric"
999
+ ],
1000
+ "limitations": [
1001
+ "professional_population_not_students",
1002
+ "single_task_ecology",
1003
+ "preprint_not_peer_reviewed"
1004
+ ],
1005
+ "confounders": [
1006
+ "task_familiarity",
1007
+ "platform_recruitment_self_selection"
1008
+ ],
1009
+ "source_location": "https://doi.org/10.48550/arXiv.2302.06590",
1010
+ "quality_dimensions": {
1011
+ "D1_study_design": 2,
1012
+ "D2_sample_quality": 2,
1013
+ "D3_measurement_validity": 2,
1014
+ "D4_temporal_strength": 1,
1015
+ "D5_directness": 1
1016
+ },
1017
+ "quality_score": 8.0,
1018
+ "evidence_level": "moderate",
1019
+ "applicability": {
1020
+ "learner_match": "mismatch_professional_developers",
1021
+ "subject_match": "adjacent_web_development_task",
1022
+ "tool_match": "copilot_like_generative_ai",
1023
+ "scope": "task_performance_only_no_learning_outcome"
1024
+ },
1025
+ "confidence": 0.6,
1026
+ "status": "SUPPORTED",
1027
+ "decision_relation": "conditional",
1028
+ "claim_id": "C-008"
1029
+ },
1030
+ {
1031
+ "evidence_id": "E-009",
1032
+ "source_id": "S-2023-yetistiren",
1033
+ "title": "GitHub Copilot AI Pair Programmer: Asset or Liability?",
1034
+ "year": 2023,
1035
+ "study_type": "observational",
1036
+ "education_level": "not_applicable_code_artifacts",
1037
+ "subject": "code_generation_benchmarks",
1038
+ "population": "取自公开基准数据集的 Copilot 生成程序与人类编写程序",
1039
+ "sample_size": null,
1040
+ "intervention": "Copilot 生成的程序",
1041
+ "comparison": "相同基准上的人类编写程序",
1042
+ "outcome_type": "code_quality",
1043
+ "outcome_measure": "correctness_security_maintainability_metrics_on_benchmarks",
1044
+ "claim": "系统性基准评估显示 Copilot 生成代码相对人类代码的质量结论不一:部分正确性具竞争力,同时记录到安全相关缺陷。",
1045
+ "direction": "neutral",
1046
+ "relation_to_claim": "neutral",
1047
+ "effect_direction": "null",
1048
+ "study_id": "STUDY-YETISTIREN-2023",
1049
+ "sample_id": "SMPL-YETISTIREN-2023-BENCH",
1050
+ "effect": "mixed quality profile; no single-direction summary",
1051
+ "duration": "not_applicable_artifact_study",
1052
+ "method": "systematic empirical evaluation of generated code against human baselines on public benchmarks",
1053
+ "strengths": [
1054
+ "multi_dimensional_quality_metrics",
1055
+ "reproducible_benchmark_protocol"
1056
+ ],
1057
+ "limitations": [
1058
+ "artifact_benchmark_not_classroom",
1059
+ "no_learning_outcome",
1060
+ "tool_version_from_2023"
1061
+ ],
1062
+ "confounders": [
1063
+ "benchmark_task_distribution"
1064
+ ],
1065
+ "source_location": "https://doi.org/10.1016/j.jss.2023.111734",
1066
+ "quality_dimensions": {
1067
+ "D1_study_design": 1,
1068
+ "D2_sample_quality": 2,
1069
+ "D3_measurement_validity": 2,
1070
+ "D4_temporal_strength": 1,
1071
+ "D5_directness": 1
1072
+ },
1073
+ "quality_score": 7.0,
1074
+ "evidence_level": "moderate",
1075
+ "applicability": {
1076
+ "learner_match": "mismatch_no_learners_in_study",
1077
+ "subject_match": "introductory_adjacent_code_tasks",
1078
+ "tool_match": "copilot_like_generative_ai",
1079
+ "scope": "output_quality_only"
1080
+ },
1081
+ "confidence": 0.55,
1082
+ "status": "SUPPORTED",
1083
+ "decision_relation": "conditional",
1084
+ "claim_id": "C-009"
1085
+ },
1086
+ {
1087
+ "evidence_id": "E-010",
1088
+ "source_id": "S-2022-finnie-ansley",
1089
+ "title": "Using GitHub Copilot to Solve Introductory Programming Problems",
1090
+ "year": 2022,
1091
+ "study_type": "observational",
1092
+ "education_level": "university_year_1_question_sets",
1093
+ "subject": "introductory_python",
1094
+ "population": "公开 CS1 考试题集,并与已发表的学生分数分布比较",
1095
+ "sample_size": null,
1096
+ "intervention": "Codex 对 CS1 题目作答生成",
1097
+ "comparison": "已发表的学生同届分数分布",
1098
+ "outcome_type": "assignment_score",
1099
+ "outcome_measure": "pass_rate_on_cs1_exam_style_questions",
1100
+ "claim": "Codex 在 CS1 考试风格题目上能给出通过水平的解答(依数据集约 50%–75%),表明新手手中存在可观的任务能力余量。",
1101
+ "direction": "support",
1102
+ "relation_to_claim": "support",
1103
+ "effect_direction": "positive",
1104
+ "study_id": "STUDY-FINNIEANSLEY-2022",
1105
+ "sample_id": "SMPL-FINNIEANSLEY-2022-QSETS",
1106
+ "effect": "passing solutions on ~50-75% of questions across datasets",
1107
+ "duration": "not_applicable_capability_probe",
1108
+ "method": "capability benchmark against published student distributions; reproducible question sets",
1109
+ "strengths": [
1110
+ "public_reproducible_question_sets",
1111
+ "directly_relevant_task_domain"
1112
+ ],
1113
+ "limitations": [
1114
+ "tool_solves_task_does_not_equate_student_learning",
1115
+ "codex_2021_model_version_outdated"
1116
+ ],
1117
+ "confounders": [
1118
+ "question_leakage_into_training_data_possible"
1119
+ ],
1120
+ "source_location": "https://dl.acm.org/doi/10.1145/3545945.3569830",
1121
+ "quality_dimensions": {
1122
+ "D1_study_design": 1,
1123
+ "D2_sample_quality": 2,
1124
+ "D3_measurement_validity": 2,
1125
+ "D4_temporal_strength": 1,
1126
+ "D5_directness": 1
1127
+ },
1128
+ "quality_score": 7.0,
1129
+ "evidence_level": "moderate",
1130
+ "applicability": {
1131
+ "learner_match": "partial_measures_tool_not_students",
1132
+ "subject_match": "introductory_programming",
1133
+ "tool_match": "copilot_like_generative_ai",
1134
+ "scope": "tool_capability_headroom"
1135
+ },
1136
+ "confidence": 0.55,
1137
+ "status": "SUPPORTED",
1138
+ "decision_relation": "conditional",
1139
+ "claim_id": "C-010"
1140
+ },
1141
+ {
1142
+ "evidence_id": "E-011",
1143
+ "source_id": "S-2023-explanations-compare",
1144
+ "title": "Comparing Code Explanations Created by Students and Large Language Models",
1145
+ "year": 2023,
1146
+ "study_type": "observational",
1147
+ "education_level": "university_introductory",
1148
+ "subject": "code_explanation_scaffolding",
1149
+ "population": "同一批短程序的学生版与 LLM 版讲解的受控对比",
1150
+ "sample_size": null,
1151
+ "intervention": "LLM 生成的代码讲解",
1152
+ "comparison": "学生撰写的同题讲解",
1153
+ "outcome_type": "metacognition",
1154
+ "outcome_measure": "rated_explanation_quality_and_comprehensibility",
1155
+ "claim": "受控比较发现 LLM 生成的代码讲解与学生自撰讲解相当(部分更优),适合作为解释性支架材料,而非替代学生的解释练习。",
1156
+ "direction": "support",
1157
+ "relation_to_claim": "support",
1158
+ "effect_direction": "positive",
1159
+ "study_id": "STUDY-EXPLCOMP-2023",
1160
+ "sample_id": "SMPL-EXPLCOMP-2023-RATINGS",
1161
+ "effect": "comparable-or-better rated quality vs student explanations",
1162
+ "duration": "single_session_ratings",
1163
+ "method": "controlled comparison with blind rating of explanation pairs",
1164
+ "strengths": [
1165
+ "controlled_pairwise_comparison",
1166
+ "learning_process_relevant_construct"
1167
+ ],
1168
+ "limitations": [
1169
+ "short_term_ratings_not_learning_gains",
1170
+ "small_program_snippets_ecology"
1171
+ ],
1172
+ "confounders": [
1173
+ "rating_criteria_subjectivity"
1174
+ ],
1175
+ "source_location": "https://dl.acm.org/doi/10.1145/3587102.3588785",
1176
+ "quality_dimensions": {
1177
+ "D1_study_design": 1,
1178
+ "D2_sample_quality": 2,
1179
+ "D3_measurement_validity": 2,
1180
+ "D4_temporal_strength": 1,
1181
+ "D5_directness": 1
1182
+ },
1183
+ "quality_score": 7.0,
1184
+ "evidence_level": "moderate",
1185
+ "applicability": {
1186
+ "learner_match": "partial_scaffold_material_only",
1187
+ "subject_match": "introductory_programming",
1188
+ "tool_match": "llm_explanations",
1189
+ "scope": "scaffold_quality_not_effectiveness"
1190
+ },
1191
+ "confidence": 0.55,
1192
+ "status": "SUPPORTED",
1193
+ "decision_relation": "conditional",
1194
+ "claim_id": "C-011"
1195
+ },
1196
+ {
1197
+ "evidence_id": "E-012",
1198
+ "source_id": "S-2022-vaithilingam",
1199
+ "title": "Expectation vs. Experience: Evaluating the Usability of Code Generation Tools",
1200
+ "year": 2022,
1201
+ "study_type": "qualitative",
1202
+ "education_level": "mixed_cs_students_and_professionals",
1203
+ "subject": "programmer_usability_of_codegen_tools",
1204
+ "population": "24 名参与者参与的组内设计可用性研究",
1205
+ "sample_size": 24,
1206
+ "intervention": "Copilot 类工具辅助编程",
1207
+ "comparison": "不使用工具的组内基线",
1208
+ "outcome_type": "over_reliance",
1209
+ "outcome_measure": "understanding_ownership_and_debugging_reports",
1210
+ "claim": "尽管首任务完成更快,参与者难以理解并调试 AI 生成的解法、对最终程序所有权感低——记录了纯速度指标遗漏的认知与依赖风险。",
1211
+ "direction": "contradict",
1212
+ "relation_to_claim": "contradict",
1213
+ "effect_direction": "negative",
1214
+ "study_id": "STUDY-VAITHILINGAM-2022",
1215
+ "sample_id": "SMPL-VAITHILINGAM-2022-N24",
1216
+ "effect": "documented comprehension/ownership difficulties despite speed gain",
1217
+ "duration": "single_session",
1218
+ "method": "within-subject usability study with tasks, observation and interviews",
1219
+ "strengths": [
1220
+ "rich_qualitative_process_data",
1221
+ "constructs_missed_by_speed_metrics"
1222
+ ],
1223
+ "limitations": [
1224
+ "small_n_24",
1225
+ "single_session",
1226
+ "self_reported_understanding"
1227
+ ],
1228
+ "confounders": [
1229
+ "participant_ai_familiarity"
1230
+ ],
1231
+ "source_location": "https://dl.acm.org/doi/10.1145/3491101.3519665",
1232
+ "quality_dimensions": {
1233
+ "D1_study_design": 1,
1234
+ "D2_sample_quality": 2,
1235
+ "D3_measurement_validity": 2,
1236
+ "D4_temporal_strength": 1,
1237
+ "D5_directness": 1
1238
+ },
1239
+ "quality_score": 7.0,
1240
+ "evidence_level": "moderate",
1241
+ "applicability": {
1242
+ "learner_match": "partial_includes_cs_students",
1243
+ "subject_match": "programming_adjacent",
1244
+ "tool_match": "copilot_like_generative_ai",
1245
+ "scope": "risk_identification"
1246
+ },
1247
+ "confidence": 0.55,
1248
+ "status": "CONTRADICT",
1249
+ "decision_relation": "conditional",
1250
+ "claim_id": "C-012"
1251
+ }
1252
+ ],
1253
+ "methodology_reviews": [
1254
+ {
1255
+ "target": "overall",
1256
+ "audit_items": {
1257
+ "control_group": {
1258
+ "status": "met",
1259
+ "note": "证据集中所有提出因果主张的量化研究均含对照条件:Bastani(E-001/E-002/E-003)为三臂 RCT(无护栏 GPT Base / 护栏 GPT Tutor / 无 AI 对照组,N=2848);Kazemitabaar(E-004/E-005)为随机对照(有/无 Codex,n=69);Shihab(E-011/E-012/E-013)为组内对照(同一批学生有/无 Copilot,n=10)。无对照的仅为不承担因果主张的质性/观察研究(Marzuki E-006/E-007 n=3 案例、Prather E-008/E-009 21 次观察性会话)与综述(Denny E-010),不构成对因果维度的威胁。但人群最匹配的实证研究(Prather、Marzuki)恰恰无对照,若下游把其主题当作效应证据则越界。"
1260
+ },
1261
+ "randomization": {
1262
+ "status": "met",
1263
+ "note": "仅最大型的两项研究实施随机分配:Bastani(课堂内随机分组,约 1000 名学生,E-001/E-002/E-003)与 Kazemitabaar(随机对照,n=69,E-004/E-005)。Shihab(E-011/E-012/E-013)为组内设计、无随机分配,叠加任务顺序与学习效应;Marzuki(E-006/E-007)与 Prather(E-008/E-009)非实验设计。随机化最充分的恰是人群最不匹配的研究(高中/数学),人群最相关的研究全部无随机化。"
1264
+ },
1265
+ "pre_test": {
1266
+ "status": "met",
1267
+ "note": "Kazemitabaar 明确报告 Scratch 前测并用于异质性分析(E-005:前测高分者保持后测显著更好);Bastani 以回归对照均值(0.284)控制基线(E-001),但证据记录未详述独立前测流程与组间基线等价性检验;Shihab(E-011/E-012)以'高度相似任务'替代前测,无真实基线测量;Marzuki/Prather/Denny 不适用。期初基线控制整体不系统,正效应可能部分来自基线差异。"
1268
+ },
1269
+ "post_test": {
1270
+ "status": "met",
1271
+ "note": "Bastani 设置移除 AI 后的独立考试(E-002:GPT Base 组 -17% 显著;E-003:Tutor 组 -0.004 不显著),是全集中设计最规范的后测;Kazemitabaar 有一周后延迟后测(E-005)。但 Shihab(E-011/E-012)仅在 AI 在场条件下测量任务表现,无无 AI 独立后测;Marzuki/Prather 无量化后测。正效应证据(E-001/E-004/E-011/E-012)全部缺乏'无 AI 独立后测'这一关键对照。"
1272
+ },
1273
+ "retention_test": {
1274
+ "status": "partial",
1275
+ "note": "仅 Kazemitabaar 提供真正的延迟保持力测量(一周后评估后测,结果不显著,E-005);Bastani 的独立考试紧随移除 AI 之后(即时测量,E-002/E-003),测的是近迁移而非学期级保持;Shihab、Marzuki、Prather 均无保持力数据。无任何研究覆盖与试点 16 周学期相当的时间尺度,长期保持(乃至期末、后续课程)证据完全空白。"
1276
+ },
1277
+ "transfer_test": {
1278
+ "status": "partial",
1279
+ "note": "最有价值的迁移形式——从'AI 在场练习'迁移到'无 AI 独立表现'——在 Bastani(E-002/E-003)与 Kazemitabaar(E-005)中得到测量,这是试点成功判据(期末无 AI 机试/笔试)的直接对应证据。但两研究迁移距离有限(同领域同题型),无跨任务类型、跨课程(如数据结构)迁移测量;Shihab(E-011/E-012)在同一环境下切换工具条件,未测迁移;frame 的 secondary 追踪(后续课程表现)尚无证据基支撑其预期方向。"
1280
+ },
1281
+ "sample_bias": {
1282
+ "status": "met",
1283
+ "note": "样本构成与目标人群(国内大一零基础 C 语言新生)系统性错位:Bastani(E-001~E-003)为土耳其 9-11 年级高中生数学;Kazemitabaar(E-004/E-005)为 10-17 岁 K12 编程新手;Marzuki(E-006/E-007)为 EFL 学术写作;唯一大学本科生样本 Shihab(E-011~E-013)仅 n=10 且任务为 brownfield Web 代码库(非零基础 C 入门);Prather(E-008/E-009)面向入门课程新手但为观察性研究。外推需跨'高中→大学'与'数学/Web→C 入门'两个维度,无单一研究同时匹配领域与年龄层。"
1284
+ },
1285
+ "self_selection": {
1286
+ "status": "partial",
1287
+ "note": "Bastani 为课堂内随机分组(E-001~E-003),自选偏倚最小,是全集最干净的样本;Kazemitabaar(E-004/E-005)系招募参加课后项目,存在一定自选;Shihab(E-011~E-013)n=10 全部自愿报名,志愿者通常更主动、技术接受度更高,正向速度/进展效应可能被高估;Marzuki(E-006/E-007)为目的性抽样/志愿者访谈样本,'珍视 AI 辅助价值'主题不能外推。自选偏倚与研究人群相关性成反比:越相关的样本自选越强。"
1288
+ },
1289
+ "measurement_validity": {
1290
+ "status": "partial",
1291
+ "note": "完成率/正确率/速度作为'学习'代理的效度可疑:E-001 练习正确率、E-004 任务得分、E-011/E-012 完成速度与进展均在 AI 可访问条件下测得,AI 可直接产出答案抬高指标,无法区分'学会了'与'抄到了'(Bastani 机制数据 E-002:GPT Base 答对率 51% 中 42% 为逻辑错误;Wermelinger S-2023 显示 Copilot 可首次尝试解决 24 道典型入门题中的 16 道,FETCH_PARTIAL 仅验证到机构库摘要)。自评测量不可靠(E-002 学生过度乐观、E-008 能力错觉)。效度较高的测量(无 AI 独立考试 E-002/E-003、延迟后测 E-005)恰恰给出负向或零结果——测量选择本身决定结论方向。"
1292
+ },
1293
+ "confounders": {
1294
+ "status": "partial",
1295
+ "note": "Bastani(E-001~E-003)三臂 RCT 控制最好(同一课堂环境、固定环节时长),但仍存在动机/参与度差异与护栏组教师设计提示带来的额外教学投入混淆。Shihab(E-011/E-012)组内设计叠加顺序效应、学习效应与疲劳('高度相似任务'第二次完成必然更快);Kazemitabaar(E-004/E-005)课后项目环境存在练习量/监督差异。各研究均未系统控制练习时间、动机与同伴效应。"
1296
+ },
1297
+ "instructor_effect": {
1298
+ "status": "not_applicable",
1299
+ "note": "证据记录未报告任何研究的教师/助教差异控制:Bastani 的护栏条件依赖'教师设计提示'(E-003),教师提示质量与护栏效果在设计中混淆,无法区分'护栏有效'与'好教师有效';Kazemitabaar 课后项目可能单一导师;Shihab 同一研究者环境(教师恒定但样本单一)。对试点而言,护栏效果的可复制性取决于教师设计提示的能力,这一条件依赖未被任何研究分离检验。"
1300
+ },
1301
+ "novelty_effect": {
1302
+ "status": "partial",
1303
+ "note": "全部正效应来自短期干预,新奇效应不可排除:Bastani 仅 4 次 90 分钟课内环节(E-001 +48%/+127%,E-002 独立考试紧随其后)、Kazemitabaar 训练期任务(E-004)与一周后测(E-005)、Shihab 单次实验室会话(E-011/E-012)。无任何研究覆盖 16 周学期周期,新奇感消退后的长期参与度与学习效应维持无证据;skeptic 的 novelty_effect 检查亦为 medium 风险。"
1304
+ },
1305
+ "tool_version_effect": {
1306
+ "status": "not_applicable",
1307
+ "note": "证据工具代际与试点工具不同代:Kazemitabaar(E-004/E-005)为 2023 年 OpenAI Codex(code-davinci 时代)、Marzuki(E-006/E-007)为 GPT-3.5/4 时代 ChatGPT、Bastani(E-001~E-003)为 GPT-4 时代、Shihab(E-011~E-013)为 2024-2025 Copilot;试点学年(2025-2026)学生面对 GPT-5 级模型与 IDE 深度集成智能体,能力更强、集成更深,收益与依赖风险可能同时放大,2023-2025 效应量不能直接外推。"
1308
+ },
1309
+ "ai_usage_policy": {
1310
+ "status": "partial",
1311
+ "note": "Bastani(E-001~E-003)是唯一直接操纵 AI 使用政策的研究:无护栏 GPT Base(类标准 ChatGPT 界面,可抄答案)vs 护栏 GPT Tutor(教师设计提示、不给直接答案),对应实证了'允许使用但无规则→独立考试 -17% 伤害'与'有护栏→练习 +127% 且负效应消除'的政策对比,与试点'禁止直接提交 AI 代码、实验课独立评测'的护栏设计同构。局限:护栏效果仅在单一情境(高中数学、教师设计提示)验证过,需在 C 语言场景复验;且各研究均未验证对照组依从性(对照组成员是否实际未使用 AI),政策污染未排除。"
1312
+ },
1313
+ "dropout": {
1314
+ "status": "partial",
1315
+ "note": "证据记录未报告各研究的退出率、缺失数据处理与依从性验证:Bastani 约 1000 名学生、4 次环节、2848 观测(E-001~E-003),跨环节流失与缺失未说明;Kazemitabaar n=69 一周后测(E-005)的保留率未报告;Shihab n=10 完成全部任务并参加退出访谈(E-011~E-013),是唯一可推断低退出的研究;Marzuki n=3 访谈样本。意图治疗 vs 实际使用(对照组偷偷用 AI)的依从性分析在全集中缺失。"
1316
+ }
1317
+ },
1318
+ "task_vs_learning_guard": {
1319
+ "measured_construct": "证据集测量了两个不同构念:(1) AI 在场时的任务表现——E-001 练习成绩(+48%/+127%)、E-004 训练期任务完成率/得分(+1.15 倍/+1.8 倍)、E-011/E-012 完成速度(快 35%)与进展(多 50%),均在 AI 可访问条件下测得,AI 可直接产出结果;(2) 移除 AI 后的独立学习结果——E-002 独立考试 -17%(显著)、E-003 独立考试 -0.004(不显著)、E-005 一周后延迟后测(null)。",
1320
+ "equates_task_with_learning": false,
1321
+ "note": "证据集本身未把任务表现等同学习效果:三类独立测量(E-002/E-003/E-005)与任务表现测量(E-001/E-004/E-011/E-012)被明确分开,且独立测量给出负向/零结果,恰是对照铁律(SKILL.md RULE 3:task performance 不得自动等同 learning effect)的正确执行。但风险在边界处:(a) 若下游综合以 E-001 的 +127% 或 E-011 的快 35% 作为'学习提升'证据,即违反铁律,证据天平会系统性偏向'允许使用';(b) E-001 练习成绩提升与 E-002 独立考试伤害在同一研究中并存,任何只引其一的做法都会误导。frame 的 outcomes 已把'任务表现'与'学习能力'分开测量、并把期末无 AI 统一机试/笔试设为唯一成功判据,与本 guard 一致。"
1322
+ },
1323
+ "verdict": "CONCERN",
1324
+ "limitations": [
1325
+ "证据集中未找到大学 C 语言课程 AI 编程助手的随机研究。",
1326
+ "任务表现增益持续较大,而学习效应估计中性偏负且学科错配。",
1327
+ "保持性证据仅有一周(Kazemitabaar 2023)。"
1328
+ ],
1329
+ "suggestions": [
1330
+ "将所有学习效应结论视为需要大学层面直接研究支撑。",
1331
+ "试点护栏设计参照 Bastani 2025 的 GPT Tutor(提示而非答案)。",
1332
+ "在评估计划中加入无 AI 迁移任务。"
1333
+ ]
1334
+ }
1335
+ ],
1336
+ "conflicts": [
1337
+ {
1338
+ "reason_for_disagreement": "分歧来自结果分离(任务 vs 学习)、工具设计(有护栏 vs 无护栏)与人群(K-12/职业者 vs 大学生)。随机实验与基准研究中任务表现证据一致为正;唯一测量移除 AI 后独立表现的研究显示无护栏时有害;可用性与工件研究补充依赖与质量警示而非解决学习问题。"
1339
+ }
1340
+ ],
1341
+ "applicability": {
1342
+ "suitable_for": "在大一 C 课程以护栏化使用政策开展试点",
1343
+ "not_suitable_for": "无使用政策的全面放开采用",
1344
+ "required_conditions": [
1345
+ "护栏化 AI 使用政策(给提示不给答案,仿 GPT Tutor 组)",
1346
+ "无 AI 迁移评估",
1347
+ "助教支持"
1348
+ ]
1349
+ },
1350
+ "intervention": {
1351
+ "decision": "pilot",
1352
+ "target_learners": "大一 C 语言编程学生(60 人讲授课班)",
1353
+ "learning_goals": [
1354
+ "在不借助 AI 的情况下独立编写并调试小型 C 程序",
1355
+ "理解核心概念:变量、条件、循环、数组、指针",
1356
+ "批判性地把 AI 用作讲解与调试辅助,而非答案机"
1357
+ ],
1358
+ "pilot_duration": "8 周",
1359
+ "phase_1": {
1360
+ "name": "阶段一 —— 独立基础",
1361
+ "activities": [
1362
+ "基线测验",
1363
+ "前 2 周作业完全不使用 AI 代码生成"
1364
+ ],
1365
+ "ai_usage_rule": "禁止完整代码生成;AI 仅可用于概念讲解",
1366
+ "outcome_check": "基线任务表现与独立问题解决测量"
1367
+ },
1368
+ "phase_2": {
1369
+ "name": "阶段二 —— 只解释,不解题",
1370
+ "activities": [
1371
+ "第 3–4 周:允许 AI 解释错误、概念与调试思路"
1372
+ ],
1373
+ "ai_usage_rule": "AI 可以解释,但不得产出完整解题方案",
1374
+ "outcome_check": "期中无 AI 测验"
1375
+ },
1376
+ "phase_3": {
1377
+ "name": "阶段三 —— 结构化协作",
1378
+ "activities": [
1379
+ "第 5–7 周:允许 AI 生成部分代码;学生必须用自己的话解释每段 AI 生成代码"
1380
+ ],
1381
+ "ai_usage_rule": "允许部分代码生成;关键逻辑须书面解释;提交需附推理痕迹",
1382
+ "outcome_check": "每周实验完成率与代码质量量表"
1383
+ },
1384
+ "phase_4": {
1385
+ "name": "阶段四 —— 迁移检验",
1386
+ "activities": [
1387
+ "第 8 周:在无 AI 环境完成新的编程任务"
1388
+ ],
1389
+ "ai_usage_rule": "迁移评估期间不得使用 AI",
1390
+ "outcome_check": "迁移测验分数、独立问题解决"
1391
+ },
1392
+ "ai_usage_policy": "AI 使用分三档明确分级(解释 / 协作 / 无 AI 迁移)。照抄未审视的 AI 输出属学术诚信违规,并通过推理痕迹要求核查。",
1393
+ "teacher_role": "设计护栏化提示与量表;监控使用日志;主持每周反思复盘",
1394
+ "student_role": "在允许模式下完成作业;提交推理痕迹;反思 AI 何时有帮助、何时掩盖了理解",
1395
+ "reflection_requirement": "学生须用自己的话解释关键 AI 生成逻辑;每阶段一份书面反思",
1396
+ "assessment": "无 AI 基线测验、期中测验、无 AI 迁移任务、代码质量量表、AI 使用自我报告",
1397
+ "risk_control": [
1398
+ "护栏化使用政策仿 Bastani 2025 GPT Tutor(提示而非答案)",
1399
+ "无 AI 迁移评估防止任务收益造成成绩虚高",
1400
+ "每周审查使用日志以发现拐杖行为"
1401
+ ],
1402
+ "stop_conditions": [
1403
+ "迁移测验成绩显著低于基线同届预期",
1404
+ "推理痕迹中出现普遍诚信违规",
1405
+ "风险指标中 AI 依赖信号超阈值",
1406
+ "助教/教师工作量不可持续"
1407
+ ],
1408
+ "evidence_alignment": [
1409
+ "E-004",
1410
+ "E-005",
1411
+ "E-006"
1412
+ ]
1413
+ },
1414
+ "evaluation": {
1415
+ "research_question": "在大一 C 课程中,护栏化的 AI 编程助手(只解释→结构化协作)相比无 AI 教学,能否在不增加 AI 依赖的前提下提升独立问题解决能力?",
1416
+ "groups": {
1417
+ "treatment": "两个采用四阶段护栏化 AI 政策的实验班",
1418
+ "comparison": "两个无 AI 的匹配对照班(同一教师与材料)"
1419
+ },
1420
+ "baseline": "第 1 周无 AI 编程测验(独立问题解决、完成用时)",
1421
+ "post_test": "第 8 周无 AI 编程测验(独立问题解决、代码质量)",
1422
+ "retention_test": "期末考试(第 16 周)—— 试点后 8 周的延迟测量",
1423
+ "transfer_test": "第 8 周严格无 AI 环境下的新编程任务",
1424
+ "process_metrics": [
1425
+ "每周实验完成率",
1426
+ "AI 使用日志:提交的提示、复制代码块、推理痕迹",
1427
+ "求助行为计数"
1428
+ ],
1429
+ "learning_metrics": [
1430
+ "独立问题解决(无 AI 测验)",
1431
+ "代码质量量表",
1432
+ "期末考试保持",
1433
+ "迁移任务得分"
1434
+ ],
1435
+ "risk_metrics": [
1436
+ "AI 依赖指数(来自推理痕迹质量的『使用而无理解』)",
1437
+ "学术诚信违规",
1438
+ "自我报告的过度依赖",
1439
+ "虚假信心(测后自信 vs 实际得分)"
1440
+ ],
1441
+ "analysis_plan": "预登记的实验班 vs 对照班基线调整学习指标比较(ANCOVA);任务表现指标与学习指标分开报告;按先验编程能力做亚组分析;第 3、5、7 周监测停止条件。",
1442
+ "success_threshold": "实验班独立问题解决非劣(差异在 5% 以内)且保持相当或更优、AI 依赖指数低于阈值;若独立问题解决下滑超过 10%,无论任务收益如何,试点均判为失败。",
1443
+ "stop_conditions": [
1444
+ "迁移测验成绩较对照班下滑超过 10%",
1445
+ "提交中诚信违规比例超过 20%",
1446
+ "AI 依赖指数连续两周超过预设阈值"
1447
+ ]
1448
+ },
1449
+ "benchmark": {},
1450
+ "provenance": {
1451
+ "search_provider": "n/a"
1452
+ }
1453
+ }