eduevidence 6.0.0 → 6.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (189) hide show
  1. package/CONTRIBUTING.md +105 -0
  2. package/README.md +93 -38
  3. package/README.zh-CN.md +26 -6
  4. package/SKILL.md +11 -2
  5. package/assets/readme/landing-tour.gif +0 -0
  6. package/assets/readme/studio-tour.gif +0 -0
  7. package/bin/eduevidence.js +2 -1
  8. package/docs/architecture.md +319 -43
  9. package/docs/demo-workplace-ai.md +1 -1
  10. package/docs/install-guide.md +1 -1
  11. package/docs/orchestration-role-model.md +1 -1
  12. package/docs/release-closeout/README.md +1 -1
  13. package/docs/sciverse-api.md +125 -0
  14. package/eduevidence_cli.py +10 -0
  15. package/engine/decision_policy.py +96 -0
  16. package/engine/evidence_graph.py +14 -10
  17. package/engine/gaps.py +42 -22
  18. package/engine/ids.py +2 -0
  19. package/engine/library.py +6 -2
  20. package/engine/living.py +34 -4
  21. package/engine/migration.py +88 -3
  22. package/engine/orchestration.py +5 -5
  23. package/engine/paths.py +2 -0
  24. package/engine/pilot.py +34 -32
  25. package/engine/taxonomy.py +211 -0
  26. package/engine/tribunal.py +43 -31
  27. package/engine/versions.py +1 -1
  28. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1360 -146
  29. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +3 -3
  30. package/examples/ai-coding-assistant-evidence/citation_check.json +1 -1
  31. package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
  32. package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
  33. package/examples/ai-coding-assistant-evidence/report_spec.json +23 -12
  34. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +447 -127
  35. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +447 -127
  36. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +447 -127
  37. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +447 -127
  38. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +447 -127
  39. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1360 -146
  40. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1360 -146
  41. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1360 -146
  42. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1360 -146
  43. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1360 -146
  44. package/examples/ai-coding-assistant-evidence/result.json +13 -9
  45. package/examples/ai-coding-assistant-evidence/result.zh.json +45 -41
  46. package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
  47. package/examples/ai-coding-assistant-evidence/verdict.json +6 -2
  48. package/examples/spaced-retrieval-practice/applicability.json +14 -0
  49. package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
  50. package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
  51. package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
  52. package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
  53. package/examples/spaced-retrieval-practice/frame.json +58 -0
  54. package/examples/spaced-retrieval-practice/gate_report.json +101 -0
  55. package/examples/spaced-retrieval-practice/methodology.json +78 -0
  56. package/examples/spaced-retrieval-practice/report_spec.json +212 -0
  57. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
  58. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
  59. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
  60. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
  61. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
  62. package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
  63. package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
  64. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
  65. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
  66. package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
  67. package/examples/spaced-retrieval-practice/result.json +942 -0
  68. package/examples/spaced-retrieval-practice/result.zh.json +942 -0
  69. package/examples/spaced-retrieval-practice/skeptic.json +70 -0
  70. package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
  71. package/examples/spaced-retrieval-practice/verdict.json +93 -0
  72. package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
  73. package/examples/workplace-ai-assistant/claims.jsonl +4 -4
  74. package/examples/workplace-ai-assistant/evidence.jsonl +4 -4
  75. package/examples/workplace-ai-assistant/evidence_graph.json +15 -15
  76. package/examples/workplace-ai-assistant/final_verdict.json +78 -0
  77. package/examples/workplace-ai-assistant/gate_report.json +101 -0
  78. package/examples/workplace-ai-assistant/report_spec.json +209 -40
  79. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +435 -105
  80. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +435 -105
  81. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +435 -105
  82. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +435 -105
  83. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +435 -105
  84. package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
  85. package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
  86. package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
  87. package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
  88. package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
  89. package/examples/workplace-ai-assistant/result.json +82 -20
  90. package/examples/workplace-ai-assistant/result.zh.json +82 -20
  91. package/examples/workplace-ai-assistant/skeptic.json +72 -0
  92. package/examples/workplace-ai-assistant/verdict.json +36 -10
  93. package/integrations/agent_mcp.py +2 -2
  94. package/package.json +12 -3
  95. package/pyproject.toml +4 -3
  96. package/references/report-copy-style.md +67 -0
  97. package/references/retrieval-compliance.md +75 -0
  98. package/references/retrieval-protocol.md +20 -0
  99. package/retrieval/audit.py +27 -3
  100. package/retrieval/fetch.py +96 -0
  101. package/retrieval/sciverse.py +398 -0
  102. package/retrieval/search.py +47 -7
  103. package/schemas/applicability.schema.json +94 -0
  104. package/schemas/chart-spec.schema.json +10 -3
  105. package/schemas/evidence.schema.json +316 -43
  106. package/schemas/fetch-result.schema.json +2 -1
  107. package/schemas/report-result.schema.json +3 -3
  108. package/schemas/report-spec.schema.json +98 -100
  109. package/schemas/skeptic.schema.json +86 -0
  110. package/schemas/source.schema.json +21 -2
  111. package/schemas/v2/finding.schema.json +5 -1
  112. package/schemas/v2/methodology-audit.schema.json +5 -1
  113. package/schemas/v2/outcome.schema.json +28 -5
  114. package/schemas/v2/study.schema.json +5 -1
  115. package/schemas/vNext/autoevolve-session.schema.json +34 -1
  116. package/schemas/vNext/eval-snapshot.schema.json +77 -1
  117. package/schemas/vNext/execution-plan.schema.json +50 -1
  118. package/schemas/vNext/gap-priority.schema.json +54 -1
  119. package/schemas/vNext/negative-search-record.schema.json +68 -1
  120. package/schemas/vNext/research-iteration.schema.json +87 -1
  121. package/schemas/vNext/research-strategy.schema.json +62 -1
  122. package/schemas/vNext/skill-experiment.schema.json +90 -1
  123. package/schemas/vNext/task-spec.schema.json +156 -1
  124. package/schemas/vNext/worker-result.schema.json +60 -1
  125. package/schemas/verdict.schema.json +164 -28
  126. package/scripts/build_esl_artifacts.py +2 -2
  127. package/scripts/build_report_variants.py +18 -2
  128. package/scripts/build_result.py +74 -9
  129. package/scripts/check_package_parity.py +85 -0
  130. package/scripts/check_protocol_alignment.py +375 -0
  131. package/scripts/check_versioned_schemas.py +254 -0
  132. package/scripts/claim_audit.py +13 -8
  133. package/scripts/compute_confidence.py +10 -0
  134. package/scripts/did_regression.py +12 -2
  135. package/scripts/evidence_score.py +5 -2
  136. package/scripts/generate_new_projects.py +4 -4
  137. package/scripts/orchestrator.py +120 -24
  138. package/scripts/pre_verdict_gate.py +224 -26
  139. package/scripts/quickstart.py +18 -2
  140. package/scripts/run_workspace.py +7 -1
  141. package/scripts/skill_payload.py +4 -1
  142. package/scripts/test_adversarial_empirical.py +26 -19
  143. package/scripts/validate_schema.py +31 -1
  144. package/skill/agents/evaluation-designer.md +20 -4
  145. package/skill/agents/evidence-analyst.md +19 -3
  146. package/skill/agents/evidence-judge.md +50 -2
  147. package/skill/agents/evidence-retriever.md +20 -3
  148. package/skill/agents/intervention-designer.md +20 -4
  149. package/skill/agents/method-reviewer.md +18 -2
  150. package/skill/agents/{education-planner.md → research-planner.md} +19 -3
  151. package/skill/agents/skeptic.md +18 -2
  152. package/skill/roles/registry.yaml +11 -11
  153. package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
  154. package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
  155. package/skill/sub-skills/data-analysis/SKILL.md +34 -15
  156. package/skill/sub-skills/ethics-review/SKILL.md +33 -10
  157. package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
  158. package/skill/sub-skills/evidence-review/SKILL.md +31 -12
  159. package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
  160. package/skill/sub-skills/literature-review/SKILL.md +35 -14
  161. package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
  162. package/skill/sub-skills/report-generation/SKILL.md +28 -0
  163. package/skill/sub-skills/research-planning/SKILL.md +41 -14
  164. package/skill/sub-skills/study-design/SKILL.md +30 -9
  165. package/skill/task-briefs/adjudicate.md +32 -7
  166. package/skill/task-briefs/applicability.md +37 -2
  167. package/skill/task-briefs/audit.md +32 -7
  168. package/skill/task-briefs/challenge.md +34 -5
  169. package/skill/task-briefs/evaluate.md +30 -5
  170. package/skill/task-briefs/extract.md +31 -8
  171. package/skill/task-briefs/frame.md +39 -10
  172. package/skill/task-briefs/intervene.md +32 -6
  173. package/skill/task-briefs/present.md +32 -8
  174. package/skill/task-briefs/projection.md +36 -2
  175. package/skill/task-briefs/retrieve.md +36 -6
  176. package/skill/workflows/decision-and-pilot.md +76 -1
  177. package/skill/workflows/evaluate-and-update.md +83 -0
  178. package/skill/workflows/evidence-review.md +104 -0
  179. package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
  180. package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
  181. package/visualization/eduevidence-report/scripts/build_report.py +512 -65
  182. package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
  183. package/visualization/eduevidence-report/scripts/lieflat_engine.py +349 -38
  184. package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
  185. package/web/architecture.html +14885 -0
  186. package/web/studio/assets/index-B8tkF44Q.css +1 -0
  187. package/web/studio/index.html +2 -2
  188. package/web/studio/assets/index-CzXocaGv.css +0 -1
  189. /package/web/studio/assets/{index-pa7jD7n4.js → index-CQ6Keoyc.js} +0 -0
@@ -3,7 +3,7 @@
3
3
  <head>
4
4
  <meta charset="utf-8">
5
5
  <meta http-equiv="Content-Security-Policy" content="default-src 'none'; script-src 'unsafe-inline'; style-src 'unsafe-inline'; img-src data:; font-src data:; connect-src 'none'; base-uri 'none'; object-src 'none'; form-action 'none'">
6
- <meta name="eduevidence-result-sha256" content="b498e00a88080c0645a9f0c2f4f513dc916f1ec4a2c6c69592363c99adb5dc17">
6
+ <meta name="eduevidence-result-sha256" content="3bce7a28f588643a3064c3d82eb749e3999cb18cb812b85c9ad6ecc2a731c6a6">
7
7
  <meta name="viewport" content="width=device-width, initial-scale=1">
8
8
  <title>我们准备在大学一年级 C 语言课程中允许学生使用生成式 AI 编程助手。它到底会不会提高学习效果?应该怎样引入?</title>
9
9
  <style>
@@ -1920,64 +1920,141 @@ select:focus-visible,
1920
1920
  </div>
1921
1921
  <p class="hero-rationale">任务表现的正面证据 + 有据可查的无护栏风险 + 混合的质量/可用性信号 + 大学层面学习证据缺失 → 有界、护栏化、带评估的试点,而非全面采用。</p>
1922
1922
  <div class="hero-insights">
1923
- <article class="hero-insight support"><span>最强支持结论</span><p class="hero-insight-text">AI 编程助手在训练期提升新手任务表现。</p></article>
1924
- <article class="hero-insight uncertain"><span>关键不确定性 / 反例</span><p class="hero-insight-text">AI 编程助手能否真正改善或保持大学新手的编程学习——本证据集中没有大学层面的直接 RCT [无直接证据]</p></article>
1925
- <article class="hero-insight risk"><span>主要风险</span><p class="hero-insight-text">无护栏使用的 AI 依赖与过度依赖风险真实且有记录(E-004),可用性发现亦予印证(E-012)。</p></article>
1926
- <article class="hero-insight next"><span>下一步</span><p class="hero-insight-text">当前结果未提供此项信息。</p></article>
1923
+ <article class="hero-insight support"><span>最强支持结论</span><p class="hero-insight-text">AI 编程助手在训练期稳定提升练习效率:69 名新手的随机对照中完成率 1.15 倍、用时 0.57 倍。</p></article>
1924
+ <article class="hero-insight uncertain"><span>关键不确定性 / 反例</span><p class="hero-insight-text">缺少大学层面的直接学习证据;唯一大规模试验显示,无护栏使用 GPT-4 的学生独立考试成绩下降 17%。</p></article>
1925
+ <article class="hero-insight risk"><span>主要风险</span><p class="hero-insight-text">无护栏使用会抬高练习表现却压低独立考试表现,而学习者往往意识不到这一落差。</p></article>
1926
+ <article class="hero-insight next"><span>下一步</span><p class="hero-insight-text">开展分阶段 CS1 试点:给提示而非答案、每周实验课使用、并以无 AI 迁移考试作为可叫停的验收条件。</p></article>
1927
1927
  </div>
1928
1928
  <p class="hero-provenance"><span>证据 / 来源</span> · 12 / 8</p>
1929
- </div></div></section><section class="brief-block brief-outcomes" id="brief-outcomes-zh"><header class="brief-block-header"><h2>任务表现 ≠ 学习效果</h2><p>只展示真正有解释力的结果分离;正向、负向与零效应按 effect_direction 编码。</p></header><div class="brief-block-body"><div class="outcome-separation" data-visual="outcome-separation"><div class="visual-heading"><h3>结果分离 · 任务表现 ≠ 学习效果</h3><p>将不同结果类型分开裁决,避免把训练时更快、更高分直接等同于真正学会。</p><p class="semantic-note">效应方向来自 evidence.effect_direction;“支持某个主张”不等于“结果是正向”。</p></div><div class="outcome-groups"><article class="outcome-group outcome-task"><h3>任务 / 近端表现</h3><ul><li><strong>完成时间</strong><span class="outcome-states"><span class="dir pos">正向效应 2</span></span></li><li><strong>代码质量</strong><span class="outcome-states"><span class="dir neu">零效应 1</span></span></li><li><strong>作业成绩</strong><span class="outcome-states"><span class="dir pos">正向效应 2</span></span></li></ul></article><article class="outcome-group outcome-learning"><h3>学习 / 保持 / 迁移</h3><ul><li><strong>知识获得</strong><span class="outcome-states"><span class="dir pos">正向效应 1</span></span></li><li><strong>记忆保持</strong><span class="outcome-states"><span class="dir neu">零效应 1</span></span></li><li><strong>独立问题解决</strong><span class="outcome-states"><span class="dir neg">负向效应 1</span><span class="dir neu">零效应 2</span></span></li></ul></article><article class="outcome-group outcome-risk"><h3>风险 / 依赖</h3><ul><li><strong>过度依赖</strong><span class="outcome-states"><span class="dir neg">负向效应 1</span></span></li></ul></article><article class="outcome-group outcome-other"><h3>其他结果</h3><ul><li><strong>元认知</strong><span class="outcome-states"><span class="dir pos">正向效应 1</span></span></li></ul></article></div></div><div class="visual-surface brief-chart" data-visual="outcome-evidence-balance"><svg viewBox="0 0 720 300" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="各结果类型证据效应分布"><title>各结果类型证据效应分布</title><desc>各结果类型的正向 / 负向 / 零效应证据条数(基于 effect_direction,不等同于 Claim 是否被支持)。</desc><rect x="0" y="0" width="720" height="300" fill="#FFFFFF"/><line x1="415.0" y1="46" x2="415.0" y2="182" stroke="#999" stroke-width="1" stroke-dasharray="3,3"/><text x="142" y="58.0" text-anchor="end" font-size="11" fill="#333">知识获得</text><rect x="415.0" y="49.8" width="132.5" height="9.4" fill="#5E8A6A"/><text x="481.2" y="58.0" text-anchor="middle" font-size="9" fill="#fff">1</text><text x="142" y="75.0" text-anchor="end" font-size="11" fill="#333">记忆保持</text><text x="142" y="92.0" text-anchor="end" font-size="11" fill="#333">独立问题解决</text><rect x="282.5" y="83.8" width="132.5" height="9.4" fill="#A85B53"/><text x="348.8" y="92.0" text-anchor="middle" font-size="9" fill="#fff">-1</text><text x="142" y="109.0" text-anchor="end" font-size="11" fill="#333">完成时间</text><rect x="415.0" y="100.8" width="265.0" height="9.4" fill="#5E8A6A"/><text x="547.5" y="109.0" text-anchor="middle" font-size="9" fill="#fff">2</text><text x="142" y="126.0" text-anchor="end" font-size="11" fill="#333">代码质量</text><text x="142" y="143.0" text-anchor="end" font-size="11" fill="#333">作业成绩</text><rect x="415.0" y="134.8" width="265.0" height="9.4" fill="#5E8A6A"/><text x="547.5" y="143.0" text-anchor="middle" font-size="9" fill="#fff">2</text><text x="142" y="160.0" text-anchor="end" font-size="11" fill="#333">元认知</text><rect x="415.0" y="151.8" width="132.5" height="9.4" fill="#5E8A6A"/><text x="481.2" y="160.0" text-anchor="middle" font-size="9" fill="#fff">1</text><text x="142" y="177.0" text-anchor="end" font-size="11" fill="#333">过度依赖</text><rect x="282.5" y="168.8" width="132.5" height="9.4" fill="#A85B53"/><text x="348.8" y="177.0" text-anchor="middle" font-size="9" fill="#fff">-1</text><rect x="415.0" y="206.1" width="8.0" height="6" fill="#C99A4A"/><rect x="415.0" y="214.9" width="8.0" height="6" fill="#C99A4A"/><rect x="415.0" y="232.4" width="8.0" height="6" fill="#C99A4A"/><rect x="150" y="14" width="10" height="10" fill="#5E8A6A"/><text x="164" y="23" font-size="10" fill="#333">正向效应</text><rect x="226" y="14" width="10" height="10" fill="#A85B53"/><text x="240" y="23" font-size="10" fill="#333">负向效应</text><rect x="302" y="14" width="10" height="10" fill="#C99A4A"/><text x="316" y="23" font-size="10" fill="#333">零效应</text></svg><p class="chart-interpretation"><strong>这意味着什么:</strong>各 Outcome 的正向 / 负向 / 零效应证据条数对比。这里展示的是 effect_direction,不是“这条证据是否支持某个主张”。</p></div></div></section><section class="brief-block brief-tribunal" id="brief-tribunal-zh"><header class="brief-block-header"><h2>证据裁决</h2><p>支持、不确定、被反驳与缺失证据分开放置,不把长段落平铺在同一层。</p></header><div class="brief-block-body"><div class="evidence-tribunal" data-visual="evidence-tribunal-grid"><p class="tribunal-summary"><strong>决策:</strong>试点验证 · <strong>置信度:</strong>中</p><div class="tribunal-grid"><article class="tribunal-card supported"><header><span aria-hidden="true">✓</span><h3>可以主张</h3><span class="tribunal-count">7</span></header><ul><li><p>AI 编程助手在训练期间可靠地提升任务表现(完成速度、正确性)—— E-001、E-006。</p><div class="tribunal-evidence-refs"><code>E-001</code><code>E-006</code></div></li><li><p>无护栏的生成式 AI 访问在移除工具后可能损害独立问题解决能力 —— E-004。</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li><li><p>护栏设计(给提示而非给答案)能大幅缓解负面学习效应 —— E-005。</p><div class="tribunal-evidence-refs"><code>E-005</code></div></li></ul><details class="tribunal-more"><summary>查看其余 4 条</summary><ul><li><p>任务表现提升并不自动等于学习提升 —— E-004 与 E-006 的研究内对照。</p><div class="tribunal-evidence-refs"><code>E-004</code><code>E-006</code></div></li><li><p>工具能力可观:Codex 能解出约半数至四分之三的 CS1 考试风格题目 —— E-010。</p><div class="tribunal-evidence-refs"><code>E-010</code></div></li><li><p>职业开发者 RCT 显示 Copilot 带来约 55% 任务提速;但职业人群限制直接性 —— E-008。</p><div class="tribunal-evidence-refs"><code>E-008</code></div></li><li><p>LLM 代码讲解的质量评级与学生自撰讲解相当,可作支架材料 —— E-011。</p><div class="tribunal-evidence-refs"><code>E-011</code></div></li></ul></details></article><article class="tribunal-card uncertain"><header><span aria-hidden="true">?</span><h3>尚不能主张</h3><span class="tribunal-count">4</span></header><ul><li><p>AI 编程助手能否真正改善或保持大学新手的编程学习——本证据集中没有大学层面的直接 RCT [无直接证据]</p></li><li><p>Kazemitabaar 2023 的一周中性保持性能否延伸到一个学期 —— E-003。</p><div class="tribunal-evidence-refs"><code>E-003</code></div></li><li><p>基准质量结论(E-009)与讲解质量评级(E-011)能否转化为课堂学习收益。</p><div class="tribunal-evidence-refs"><code>E-009</code><code>E-011</code></div></li></ul><details class="tribunal-more"><summary>查看其余 1 条</summary><ul><li><p>可用性研究所记录的理解/所有权困难(E-012)在整学期护栏条件下会如何演变。</p><div class="tribunal-evidence-refs"><code>E-012</code></div></li></ul></details></article><article class="tribunal-card contradicted"><header><span aria-hidden="true">×</span><h3>被反驳的主张</h3><span class="tribunal-count">2</span></header><ul><li><p>&#x27;AI 工具总能提高学习&#x27;被 E-004 反驳(无护栏访问,独立考试 −17%)。</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li><li><p>&#x27;速度收益等于学习收益&#x27;被 E-001/E-006/E-008 与 E-004 之间的任务-学习分离所反驳。</p><div class="tribunal-evidence-refs"><code>E-001</code><code>E-006</code><code>E-008</code><code>E-004</code></div></li></ul></article><article class="tribunal-card missing"><header><span aria-hidden="true">…</span><h3>缺失证据</h3><span class="tribunal-count">4</span></header><ul><li><p>在大学编程课程中带保持与无 AI 迁移测试的 RCT。</p></li><li><p>同一课程内变化 AI 使用政策的研究。</p></li><li><p>跨越一门课的 AI 依赖纵向数据。</p></li></ul><details class="tribunal-more"><summary>查看其余 1 条</summary><ul><li><p>职业提速 RCT 的同行评审重复(Peng 等仍为预印本)。</p></li></ul></details></article></div></div></div></section><section class="brief-block brief-action" id="brief-action-zh"><header class="brief-block-header"><h2>从证据到行动</h2><p>适用性、护栏、停止条件与评价连成一条可执行路径。</p></header><div class="brief-block-body"><div class="evidence-to-action" data-visual="evidence-to-action"><article class="action-node action-evidence"><span>证据</span><p class="action-node-text">AI 编程助手在训练期间可靠地提升任务表现(完成速度、正确性)—— E-001、E-006。</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-applicability"><span>适用性</span><p class="action-node-text">在大一 C 课程以护栏化使用政策开展试点</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-decision"><span>决策</span><p class="action-node-text">试点验证</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-guardrails"><span>护栏</span><p class="action-node-text">AI 使用分三档明确分级(解释 / 协作 / 无 AI 迁移)。照抄未审视的 AI 输出属学术诚信违规,并通过推理痕迹要求核查。</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-stop"><span>停止条件</span><p class="action-node-text">迁移测验成绩显著低于基线同届预期; 推理痕迹中出现普遍诚信违规; 风险指标中 AI 依赖信号超阈值; 助教/教师工作量不可持续</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-evaluation"><span>评价</span><p class="action-node-text">实验班独立问题解决非劣(差异在 5% 以内)且保持相当或更优、AI 依赖指数低于阈值;若独立问题解决下滑超过 10%,无论任务收益如何,试点均判为失败。</p></article></div></div></section><section class="brief-block brief-lieflat" id="brief-lieflat-zh"><header class="brief-block-header"><h2>Lieflat 实证手作画廊</h2><p>AI 按数据形状从 Lieflat 目录选型编排;每张图的数字都可溯源到 result.json。</p></header><div class="brief-block-body"><div class="lieflat-gallery-container"><figure class="lieflat-card" data-lieflat data-visual="lieflat-bubble_almanac" data-chart-id="lieflat-bubble-almanac.svg"><h3 class="lieflat-title">发表年份 × 结果维度文献年历</h3><p class="lieflat-sub">气泡面积 ∝ 该格研究数(sqrt 换算) · 实心圆 = 有显著结果</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 540 300" width="100%" height="100%" role="img" aria-label="发表年份 × 结果维度文献年历" style="background:#111827;">
1930
- <line x1="44" y1="70" x2="520" y2="70" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:0ms"/>
1931
- <line x1="44" y1="77" x2="520" y2="77" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:14ms"/>
1932
- <line x1="44" y1="84" x2="520" y2="84" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:28ms"/>
1933
- <line x1="44" y1="91" x2="520" y2="91" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:42ms"/>
1934
- <line x1="44" y1="98" x2="520" y2="98" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:56ms"/>
1935
- <line x1="44" y1="105" x2="520" y2="105" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:70ms"/>
1936
- <line x1="44" y1="112" x2="520" y2="112" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:84ms"/>
1937
- <line x1="44" y1="119" x2="520" y2="119" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:98ms"/>
1938
- <line x1="44" y1="126" x2="520" y2="126" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:112ms"/>
1939
- <line x1="44" y1="133" x2="520" y2="133" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:126ms"/>
1940
- <line x1="44" y1="140" x2="520" y2="140" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:140ms"/>
1941
- <line x1="44" y1="147" x2="520" y2="147" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:154ms"/>
1942
- <line x1="44" y1="154" x2="520" y2="154" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:168ms"/>
1943
- <line x1="44" y1="161" x2="520" y2="161" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:182ms"/>
1944
- <line x1="44" y1="168" x2="520" y2="168" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:196ms"/>
1945
- <line x1="44" y1="175" x2="520" y2="175" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:210ms"/>
1946
- <line x1="44" y1="182" x2="520" y2="182" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:224ms"/>
1947
- <line x1="44" y1="189" x2="520" y2="189" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:238ms"/>
1948
- <line x1="44" y1="196" x2="520" y2="196" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:252ms"/>
1949
- <line x1="44" y1="203" x2="520" y2="203" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:266ms"/>
1950
- <line x1="44" y1="210" x2="520" y2="210" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:280ms"/>
1951
- <line x1="44" y1="217" x2="520" y2="217" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:294ms"/>
1952
- <line x1="44" y1="224" x2="520" y2="224" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:308ms"/>
1953
- <line x1="44" y1="231" x2="520" y2="231" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:322ms"/>
1954
- <line x1="44" y1="238" x2="520" y2="238" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:336ms"/>
1955
- <line x1="44" y1="245" x2="520" y2="245" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:350ms"/>
1956
- <line x1="44" y1="252" x2="520" y2="252" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:364ms"/>
1957
- <line x1="44" y1="259" x2="520" y2="259" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:378ms"/>
1958
- <text x="150" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:40ms">completion tim</text>
1959
- <text x="203" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:140ms">independent pr</text>
1960
- <text x="256" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:240ms">retention</text>
1961
- <text x="309" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:340ms">assignment sco</text>
1962
- <text x="361" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:440ms">knowledge gain</text>
1963
- <text x="414" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:540ms">code quality</text>
1964
- <text x="467" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:640ms">metacognition</text>
1965
- <text x="520" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:740ms">over reliance</text>
1966
- <text x="96" y="96" fill="#94A3B8" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:80ms">2022</text>
1967
- <text x="96" y="140" fill="#94A3B8" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:180ms">2023</text>
1968
- <text x="96" y="184" fill="#94A3B8" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:280ms">2024</text>
1969
- <text x="96" y="228" fill="#94A3B8" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:380ms">2025</text>
1970
- <circle cx="308.6" cy="92.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:100ms"><title>assignment_score (2022) — N = 1 studies, significant = 0</title></circle>
1971
- <circle cx="520.0" cy="92.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:112ms"><title>over_reliance (2022) — N = 1 studies, significant = 0</title></circle>
1972
- <circle cx="150.0" cy="136.0" r="5.1" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:124ms"><title>completion_time (2023) — N = 2 studies, significant = 0</title></circle>
1973
- <circle cx="202.9" cy="136.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:136ms"><title>independent_problem_solving (2023) — N = 1 studies, significant = 0</title></circle>
1974
- <circle cx="255.7" cy="136.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:148ms"><title>retention (2023) — N = 1 studies, significant = 0</title></circle>
1975
- <circle cx="414.3" cy="136.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:160ms"><title>code_quality (2023) — N = 1 studies, significant = 0</title></circle>
1976
- <circle cx="467.1" cy="136.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:172ms"><title>metacognition (2023) — N = 1 studies, significant = 0</title></circle>
1977
- <circle cx="361.4" cy="180.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:184ms"><title>knowledge_gain (2024) — N = 1 studies, significant = 0</title></circle>
1978
- <circle cx="202.9" cy="224.0" r="5.1" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:196ms"><title>independent_problem_solving (2025) — N = 2 studies, significant = 0</title></circle>
1979
- <circle cx="308.6" cy="224.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:208ms"><title>assignment_score (2025) — N = 1 studies, significant = 0</title></circle>
1980
- </svg></div><figcaption class="lieflat-caption">仅当证据集携带发表年份与结果维度时绘制。</figcaption><p class="lieflat-src">L9 Bubble Almanac · evidence.year_x_dimension</p></figure><figure class="lieflat-card" data-lieflat data-visual="lieflat-tick_rows" data-chart-id="lieflat-tick-rows.svg"><h3 class="lieflat-title">各结果类型效应方向分布</h3><p class="lieflat-sub">每 1 个圆点 = 1 条证据 · 绿 = 正向 · 灰 = 零效应 · 橙 = 负向 · 右端数字 = 净效应</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 540 300" width="100%" height="100%" role="img" aria-label="各结果类型效应方向分布" style="background:#111827;">
1929
+ </div></div></section><section class="brief-block brief-outcomes" id="brief-outcomes-zh"><header class="brief-block-header"><h2>任务表现 ≠ 学习效果</h2><p>只展示真正有解释力的结果分离;正向、负向与零效应按 effect_direction 编码。</p></header><div class="brief-block-body"><div class="outcome-separation" data-visual="outcome-separation"><div class="visual-heading"><h3>结果分离 · 任务表现 ≠ 学习效果</h3><p>将不同结果类型分开裁决,避免把训练时更快、更高分直接等同于真正学会。</p><p class="semantic-note">效应方向来自 evidence.effect_direction;“支持某个主张”不等于“结果是正向”。</p></div><div class="outcome-groups"><article class="outcome-group outcome-task"><h3>任务 / 近端表现</h3><ul><li><strong>完成时间</strong><span class="outcome-states"><span class="dir pos">正向效应 2</span></span></li><li><strong>代码质量</strong><span class="outcome-states"><span class="dir neu">零效应 1</span></span></li><li><strong>作业成绩</strong><span class="outcome-states"><span class="dir pos">正向效应 2</span></span></li></ul></article><article class="outcome-group outcome-learning"><h3>学习 / 保持 / 迁移</h3><ul><li><strong>知识获得</strong><span class="outcome-states"><span class="dir pos">正向效应 1</span></span></li><li><strong>记忆保持</strong><span class="outcome-states"><span class="dir neu">零效应 1</span></span></li><li><strong>独立问题解决</strong><span class="outcome-states"><span class="dir neg">负向效应 1</span><span class="dir neu">零效应 2</span></span></li></ul></article><article class="outcome-group outcome-risk"><h3>风险 / 依赖</h3><ul><li><strong>过度依赖</strong><span class="outcome-states"><span class="dir neg">负向效应 1</span></span></li></ul></article><article class="outcome-group outcome-other"><h3>其他结果</h3><ul><li><strong>元认知</strong><span class="outcome-states"><span class="dir pos">正向效应 1</span></span></li></ul></article></div></div><div class="visual-surface brief-chart" data-visual="outcome-evidence-balance"><svg viewBox="0 0 720 300" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="各结果类型证据效应分布"><title>各结果类型证据效应分布</title><desc>各结果类型的正向 / 负向 / 零效应证据条数(基于 effect_direction,不等同于 Claim 是否被支持)。</desc><rect x="0" y="0" width="720" height="300" fill="#FFFFFF"/><line x1="415.0" y1="46" x2="415.0" y2="182" stroke="#999" stroke-width="1" stroke-dasharray="3,3"/><text x="142" y="58.0" text-anchor="end" font-size="11" fill="#333">知识获得</text><rect x="415.0" y="49.8" width="132.5" height="9.4" fill="#5E8A6A"/><text x="481.2" y="58.0" text-anchor="middle" font-size="9" fill="#fff">1</text><text x="142" y="75.0" text-anchor="end" font-size="11" fill="#333">记忆保持</text><text x="142" y="92.0" text-anchor="end" font-size="11" fill="#333">独立问题解决</text><rect x="282.5" y="83.8" width="132.5" height="9.4" fill="#A85B53"/><text x="348.8" y="92.0" text-anchor="middle" font-size="9" fill="#fff">-1</text><text x="142" y="109.0" text-anchor="end" font-size="11" fill="#333">完成时间</text><rect x="415.0" y="100.8" width="265.0" height="9.4" fill="#5E8A6A"/><text x="547.5" y="109.0" text-anchor="middle" font-size="9" fill="#fff">2</text><text x="142" y="126.0" text-anchor="end" font-size="11" fill="#333">代码质量</text><text x="142" y="143.0" text-anchor="end" font-size="11" fill="#333">作业成绩</text><rect x="415.0" y="134.8" width="265.0" height="9.4" fill="#5E8A6A"/><text x="547.5" y="143.0" text-anchor="middle" font-size="9" fill="#fff">2</text><text x="142" y="160.0" text-anchor="end" font-size="11" fill="#333">元认知</text><rect x="415.0" y="151.8" width="132.5" height="9.4" fill="#5E8A6A"/><text x="481.2" y="160.0" text-anchor="middle" font-size="9" fill="#fff">1</text><text x="142" y="177.0" text-anchor="end" font-size="11" fill="#333">过度依赖</text><rect x="282.5" y="168.8" width="132.5" height="9.4" fill="#A85B53"/><text x="348.8" y="177.0" text-anchor="middle" font-size="9" fill="#fff">-1</text><rect x="415.0" y="206.1" width="8.0" height="6" fill="#C99A4A"/><rect x="415.0" y="214.9" width="8.0" height="6" fill="#C99A4A"/><rect x="415.0" y="232.4" width="8.0" height="6" fill="#C99A4A"/><rect x="150" y="14" width="10" height="10" fill="#5E8A6A"/><text x="164" y="23" font-size="10" fill="#333">正向效应</text><rect x="226" y="14" width="10" height="10" fill="#A85B53"/><text x="240" y="23" font-size="10" fill="#333">负向效应</text><rect x="302" y="14" width="10" height="10" fill="#C99A4A"/><text x="316" y="23" font-size="10" fill="#333">零效应</text></svg><p class="chart-interpretation"><strong>这意味着什么:</strong>各 Outcome 的正向 / 负向 / 零效应证据条数对比。这里展示的是 effect_direction,不是“这条证据是否支持某个主张”。</p></div></div></section><section class="brief-block brief-tribunal" id="brief-tribunal-zh"><header class="brief-block-header"><h2>证据裁决</h2><p>支持、不确定、被反驳与缺失证据分开放置,不把长段落平铺在同一层。</p></header><div class="brief-block-body"><div class="evidence-tribunal" data-visual="evidence-tribunal-grid"><p class="tribunal-summary"><strong>决策:</strong>试点验证 · <strong>置信度:</strong>中</p><div class="tribunal-grid"><article class="tribunal-card supported"><header><span aria-hidden="true">✓</span><h3>可以主张</h3><span class="tribunal-count">7</span></header><ul><li><p>AI 编程助手在训练期间可靠地提升任务表现(完成速度、正确性)—— E-001、E-006。</p><div class="tribunal-evidence-refs"><code>E-001</code><code>E-006</code></div></li><li><p>无护栏的生成式 AI 访问在移除工具后可能损害独立问题解决能力 —— E-004。</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li><li><p>护栏设计(给提示而非给答案)能大幅缓解负面学习效应 —— E-005。</p><div class="tribunal-evidence-refs"><code>E-005</code></div></li></ul><details class="tribunal-more"><summary>查看其余 4 条</summary><ul><li><p>任务表现提升并不自动等于学习提升 —— E-004 与 E-006 的研究内对照。</p><div class="tribunal-evidence-refs"><code>E-004</code><code>E-006</code></div></li><li><p>工具能力可观:Codex 能解出约半数至四分之三的 CS1 考试风格题目 —— E-010。</p><div class="tribunal-evidence-refs"><code>E-010</code></div></li><li><p>职业开发者 RCT 显示 Copilot 带来约 55% 任务提速;但职业人群限制直接性 —— E-008。</p><div class="tribunal-evidence-refs"><code>E-008</code></div></li><li><p>LLM 代码讲解的质量评级与学生自撰讲解相当,可作支架材料 —— E-011。</p><div class="tribunal-evidence-refs"><code>E-011</code></div></li></ul></details></article><article class="tribunal-card uncertain"><header><span aria-hidden="true">?</span><h3>尚不能主张</h3><span class="tribunal-count">4</span></header><ul><li><p>AI 编程助手能否真正改善或保持大学新手的编程学习——本证据集中没有大学层面的直接 RCT [无直接证据]</p></li><li><p>Kazemitabaar 2023 的一周中性保持性能否延伸到一个学期 —— E-003。</p><div class="tribunal-evidence-refs"><code>E-003</code></div></li><li><p>基准质量结论(E-009)与讲解质量评级(E-011)能否转化为课堂学习收益。</p><div class="tribunal-evidence-refs"><code>E-009</code><code>E-011</code></div></li></ul><details class="tribunal-more"><summary>查看其余 1 条</summary><ul><li><p>可用性研究所记录的理解/所有权困难(E-012)在整学期护栏条件下会如何演变。</p><div class="tribunal-evidence-refs"><code>E-012</code></div></li></ul></details></article><article class="tribunal-card contradicted"><header><span aria-hidden="true">×</span><h3>被反驳的主张</h3><span class="tribunal-count">2</span></header><ul><li><p>&#x27;AI 工具总能提高学习&#x27;被 E-004 反驳(无护栏访问,独立考试 −17%)。</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li><li><p>&#x27;速度收益等于学习收益&#x27;被 E-001/E-006/E-008 与 E-004 之间的任务-学习分离所反驳。</p><div class="tribunal-evidence-refs"><code>E-001</code><code>E-006</code><code>E-008</code><code>E-004</code></div></li></ul></article><article class="tribunal-card missing"><header><span aria-hidden="true">…</span><h3>缺失证据</h3><span class="tribunal-count">4</span></header><ul><li><p>在大学编程课程中带保持与无 AI 迁移测试的 RCT。</p></li><li><p>同一课程内变化 AI 使用政策的研究。</p></li><li><p>跨越一门课的 AI 依赖纵向数据。</p></li></ul><details class="tribunal-more"><summary>查看其余 1 条</summary><ul><li><p>职业提速 RCT 的同行评审重复(Peng 等仍为预印本)。</p></li></ul></details></article></div></div></div></section><section class="brief-block brief-action" id="brief-action-zh"><header class="brief-block-header"><h2>从证据到行动</h2><p>适用性、护栏、停止条件与评价连成一条可执行路径。</p></header><div class="brief-block-body"><div class="evidence-to-action" data-visual="evidence-to-action"><article class="action-node action-evidence"><span>证据</span><p class="action-node-text">AI 编程助手在训练期间可靠地提升任务表现(完成速度、正确性)—— E-001、E-006。</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-applicability"><span>适用性</span><p class="action-node-text">在大一 C 课程以护栏化使用政策开展试点</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-decision"><span>决策</span><p class="action-node-text">试点验证</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-guardrails"><span>护栏</span><p class="action-node-text">AI 使用分三档明确分级(解释 / 协作 / 无 AI 迁移)。照抄未审视的 AI 输出属学术诚信违规,并通过推理痕迹要求核查。</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-stop"><span>停止条件</span><p class="action-node-text">迁移测验成绩显著低于基线同届预期; 推理痕迹中出现普遍诚信违规; 风险指标中 AI 依赖信号超阈值; 助教/教师工作量不可持续</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-evaluation"><span>评价</span><p class="action-node-text">实验班独立问题解决非劣(差异在 5% 以内)且保持相当或更优、AI 依赖指数低于阈值;若独立问题解决下滑超过 10%,无论任务收益如何,试点均判为失败。</p></article></div></div></section><section class="brief-block brief-lieflat" id="brief-lieflat-zh"><header class="brief-block-header"><h2>Lieflat 实证手作画廊</h2><p>AI 按数据形状从 Lieflat 目录选型编排;每张图的数字都可溯源到 result.json。</p></header><div class="brief-block-body"><div class="lieflat-gallery-container"><figure class="lieflat-card" data-lieflat data-visual="lieflat-bubble_almanac" data-chart-id="lieflat-bubble-almanac.svg"><h3 class="lieflat-title">发表年份 × 结果维度文献年历</h3><p class="lieflat-sub">气泡面积 ∝ 该格研究数 · 实心圆 = 有显著结果</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 540 300" width="100%" height="100%" role="img" aria-label="发表年份 × 结果维度文献年历" style="background:#111827;">
1930
+ <line x1="44" y1="70.0" x2="520" y2="70.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:0ms"/>
1931
+ <line x1="44" y1="77.0" x2="520" y2="77.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:14ms"/>
1932
+ <line x1="44" y1="84.0" x2="520" y2="84.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:28ms"/>
1933
+ <line x1="44" y1="91.0" x2="520" y2="91.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:42ms"/>
1934
+ <line x1="44" y1="98.0" x2="520" y2="98.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:56ms"/>
1935
+ <line x1="44" y1="105.0" x2="520" y2="105.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:70ms"/>
1936
+ <line x1="44" y1="112.0" x2="520" y2="112.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:84ms"/>
1937
+ <line x1="44" y1="119.0" x2="520" y2="119.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:98ms"/>
1938
+ <line x1="44" y1="126.0" x2="520" y2="126.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:112ms"/>
1939
+ <line x1="44" y1="133.0" x2="520" y2="133.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:126ms"/>
1940
+ <line x1="44" y1="140.0" x2="520" y2="140.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:140ms"/>
1941
+ <line x1="44" y1="147.0" x2="520" y2="147.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:154ms"/>
1942
+ <line x1="44" y1="154.0" x2="520" y2="154.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:168ms"/>
1943
+ <line x1="44" y1="161.0" x2="520" y2="161.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:182ms"/>
1944
+ <line x1="44" y1="168.0" x2="520" y2="168.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:196ms"/>
1945
+ <line x1="44" y1="175.0" x2="520" y2="175.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:210ms"/>
1946
+ <line x1="44" y1="182.0" x2="520" y2="182.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:224ms"/>
1947
+ <line x1="44" y1="189.0" x2="520" y2="189.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:238ms"/>
1948
+ <line x1="44" y1="196.0" x2="520" y2="196.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:252ms"/>
1949
+ <line x1="44" y1="203.0" x2="520" y2="203.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:266ms"/>
1950
+ <line x1="44" y1="210.0" x2="520" y2="210.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:280ms"/>
1951
+ <line x1="44" y1="217.0" x2="520" y2="217.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:294ms"/>
1952
+ <line x1="44" y1="224.0" x2="520" y2="224.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:308ms"/>
1953
+ <line x1="44" y1="231.0" x2="520" y2="231.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:322ms"/>
1954
+ <line x1="44" y1="238.0" x2="520" y2="238.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:336ms"/>
1955
+ <line x1="44" y1="245.0" x2="520" y2="245.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:350ms"/>
1956
+ <line x1="44" y1="252.0" x2="520" y2="252.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:364ms"/>
1957
+ <line x1="44" y1="259.0" x2="520" y2="259.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:378ms"/>
1958
+ <text x="150" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:40ms">完成时间</text>
1959
+ <text x="203" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:140ms">独立问题解决</text>
1960
+ <text x="256" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:240ms">记忆保持</text>
1961
+ <text x="309" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:340ms">作业成绩</text>
1962
+ <text x="361" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:440ms">知识获得</text>
1963
+ <text x="414" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:540ms">代码质量</text>
1964
+ <text x="467" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:640ms">元认知</text>
1965
+ <text x="514.0" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:740ms">过度依赖</text>
1966
+ <text x="96" y="96.0" fill="#94A3B8" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:80ms">2022</text>
1967
+ <text x="96" y="140.0" fill="#94A3B8" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:180ms">2023</text>
1968
+ <text x="96" y="184.0" fill="#94A3B8" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:280ms">2024</text>
1969
+ <text x="96" y="228.0" fill="#94A3B8" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:380ms">2025</text>
1970
+ <circle cx="308.6" cy="92.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:100ms"><title>作业成绩 (2022) — N = 1 篇研究, 显著 = 0</title></circle>
1971
+ <circle cx="520.0" cy="92.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:112ms"><title>过度依赖 (2022) — N = 1 篇研究, 显著 = 0</title></circle>
1972
+ <circle cx="150.0" cy="136.0" r="5.1" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:124ms"><title>完成时间 (2023) — N = 2 篇研究, 显著 = 0</title></circle>
1973
+ <circle cx="202.9" cy="136.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:136ms"><title>独立问题解决 (2023) — N = 1 篇研究, 显著 = 0</title></circle>
1974
+ <circle cx="255.7" cy="136.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:148ms"><title>记忆保持 (2023) — N = 1 篇研究, 显著 = 0</title></circle>
1975
+ <circle cx="414.3" cy="136.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:160ms"><title>代码质量 (2023) — N = 1 篇研究, 显著 = 0</title></circle>
1976
+ <circle cx="467.1" cy="136.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:172ms"><title>元认知 (2023) — N = 1 篇研究, 显著 = 0</title></circle>
1977
+ <circle cx="361.4" cy="180.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:184ms"><title>知识获得 (2024) — N = 1 篇研究, 显著 = 0</title></circle>
1978
+ <circle cx="202.9" cy="224.0" r="5.1" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:196ms"><title>独立问题解决 (2025) — N = 2 篇研究, 显著 = 0</title></circle>
1979
+ <circle cx="308.6" cy="224.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:208ms"><title>作业成绩 (2025) — N = 1 篇研究, 显著 = 0</title></circle>
1980
+ </svg></div><figcaption class="lieflat-caption">仅当证据集携带发表年份与结果维度时绘制。</figcaption><p class="lieflat-src">L9 Bubble Almanac · Evidence.year X Dimension</p></figure><figure class="lieflat-card" data-lieflat data-visual="lieflat-matrix_heat" data-chart-id="lieflat-matrix-heat.svg"><h3 class="lieflat-title">年份 × 结果维度证据密度</h3><p class="lieflat-sub">每格数字 = 该年份该结果维度的证据条数</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 540 440" width="100%" height="100%" role="img" aria-label="年份 × 结果维度证据密度" style="background:#111827;">
1981
+ <text x="196.2" y="66" text-anchor="middle" font-size="9" font-weight="800" fill="#94A3B8" class="lf-fade" style="--motion-delay:40ms">2022</text>
1982
+ <text x="288.8" y="66" text-anchor="middle" font-size="9" font-weight="800" fill="#94A3B8" class="lf-fade" style="--motion-delay:140ms">2023</text>
1983
+ <text x="381.2" y="66" text-anchor="middle" font-size="9" font-weight="800" fill="#94A3B8" class="lf-fade" style="--motion-delay:240ms">2024</text>
1984
+ <text x="473.8" y="66" text-anchor="middle" font-size="9" font-weight="800" fill="#94A3B8" class="lf-fade" style="--motion-delay:340ms">2025</text>
1985
+ <text x="138" y="97.0" text-anchor="end" font-size="8" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:60ms">完成时间</text>
1986
+ <rect x="152.0" y="80.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:0ms"/>
1987
+ <text x="196.2" y="97.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:60ms">0</text>
1988
+ <rect x="244.5" y="80.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.90" class="lf-pop" style="--motion-delay:12ms"/>
1989
+ <text x="288.8" y="97.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:72ms">2</text>
1990
+ <rect x="337.0" y="80.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:24ms"/>
1991
+ <text x="381.2" y="97.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:84ms">0</text>
1992
+ <rect x="429.5" y="80.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:36ms"/>
1993
+ <text x="473.8" y="97.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:96ms">0</text>
1994
+ <text x="138" y="129.0" text-anchor="end" font-size="8" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:160ms">独立问题解决</text>
1995
+ <rect x="152.0" y="112.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:12ms"/>
1996
+ <text x="196.2" y="129.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:72ms">0</text>
1997
+ <rect x="244.5" y="112.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.54" class="lf-pop" style="--motion-delay:24ms"/>
1998
+ <text x="288.8" y="129.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:84ms">1</text>
1999
+ <rect x="337.0" y="112.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:36ms"/>
2000
+ <text x="381.2" y="129.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:96ms">0</text>
2001
+ <rect x="429.5" y="112.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.90" class="lf-pop" style="--motion-delay:48ms"/>
2002
+ <text x="473.8" y="129.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:108ms">2</text>
2003
+ <text x="138" y="161.0" text-anchor="end" font-size="8" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:260ms">记忆保持</text>
2004
+ <rect x="152.0" y="144.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:24ms"/>
2005
+ <text x="196.2" y="161.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:84ms">0</text>
2006
+ <rect x="244.5" y="144.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.54" class="lf-pop" style="--motion-delay:36ms"/>
2007
+ <text x="288.8" y="161.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:96ms">1</text>
2008
+ <rect x="337.0" y="144.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:48ms"/>
2009
+ <text x="381.2" y="161.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:108ms">0</text>
2010
+ <rect x="429.5" y="144.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:60ms"/>
2011
+ <text x="473.8" y="161.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:120ms">0</text>
2012
+ <text x="138" y="193.0" text-anchor="end" font-size="8" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:360ms">作业成绩</text>
2013
+ <rect x="152.0" y="176.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.54" class="lf-pop" style="--motion-delay:36ms"/>
2014
+ <text x="196.2" y="193.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:96ms">1</text>
2015
+ <rect x="244.5" y="176.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:48ms"/>
2016
+ <text x="288.8" y="193.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:108ms">0</text>
2017
+ <rect x="337.0" y="176.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:60ms"/>
2018
+ <text x="381.2" y="193.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:120ms">0</text>
2019
+ <rect x="429.5" y="176.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.54" class="lf-pop" style="--motion-delay:72ms"/>
2020
+ <text x="473.8" y="193.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:132ms">1</text>
2021
+ <text x="138" y="225.0" text-anchor="end" font-size="8" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:460ms">知识获得</text>
2022
+ <rect x="152.0" y="208.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:48ms"/>
2023
+ <text x="196.2" y="225.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:108ms">0</text>
2024
+ <rect x="244.5" y="208.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:60ms"/>
2025
+ <text x="288.8" y="225.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:120ms">0</text>
2026
+ <rect x="337.0" y="208.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.54" class="lf-pop" style="--motion-delay:72ms"/>
2027
+ <text x="381.2" y="225.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:132ms">1</text>
2028
+ <rect x="429.5" y="208.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:84ms"/>
2029
+ <text x="473.8" y="225.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:144ms">0</text>
2030
+ <text x="138" y="257.0" text-anchor="end" font-size="8" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:560ms">代码质量</text>
2031
+ <rect x="152.0" y="240.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:60ms"/>
2032
+ <text x="196.2" y="257.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:120ms">0</text>
2033
+ <rect x="244.5" y="240.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.54" class="lf-pop" style="--motion-delay:72ms"/>
2034
+ <text x="288.8" y="257.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:132ms">1</text>
2035
+ <rect x="337.0" y="240.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:84ms"/>
2036
+ <text x="381.2" y="257.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:144ms">0</text>
2037
+ <rect x="429.5" y="240.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:96ms"/>
2038
+ <text x="473.8" y="257.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:156ms">0</text>
2039
+ <text x="138" y="289.0" text-anchor="end" font-size="8" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:660ms">元认知</text>
2040
+ <rect x="152.0" y="272.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:72ms"/>
2041
+ <text x="196.2" y="289.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:132ms">0</text>
2042
+ <rect x="244.5" y="272.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.54" class="lf-pop" style="--motion-delay:84ms"/>
2043
+ <text x="288.8" y="289.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:144ms">1</text>
2044
+ <rect x="337.0" y="272.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:96ms"/>
2045
+ <text x="381.2" y="289.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:156ms">0</text>
2046
+ <rect x="429.5" y="272.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:108ms"/>
2047
+ <text x="473.8" y="289.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:168ms">0</text>
2048
+ <text x="138" y="321.0" text-anchor="end" font-size="8" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:760ms">过度依赖</text>
2049
+ <rect x="152.0" y="304.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.54" class="lf-pop" style="--motion-delay:84ms"/>
2050
+ <text x="196.2" y="321.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:144ms">1</text>
2051
+ <rect x="244.5" y="304.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:96ms"/>
2052
+ <text x="288.8" y="321.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:156ms">0</text>
2053
+ <rect x="337.0" y="304.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:108ms"/>
2054
+ <text x="381.2" y="321.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:168ms">0</text>
2055
+ <rect x="429.5" y="304.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:120ms"/>
2056
+ <text x="473.8" y="321.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:180ms">0</text>
2057
+ </svg></div><figcaption class="lieflat-caption">当证据跨多个年份与结果维度时,展示研究密度的分布。</figcaption><p class="lieflat-src">L16 Matrix Heat · Evidence.year X Outcome Counts</p></figure><figure class="lieflat-card" data-lieflat data-visual="lieflat-tick_rows" data-chart-id="lieflat-tick-rows.svg"><h3 class="lieflat-title">各结果类型效应方向分布</h3><p class="lieflat-sub">每 1 个圆点 = 1 条证据 · 绿 = 正向 · 灰 = 零效应 · 橙 = 负向</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 540 300" width="100%" height="100%" role="img" aria-label="各结果类型效应方向分布" style="background:#111827;">
1981
2058
  <text x="128" y="102.0" text-anchor="end" font-size="9" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:40ms">完成时间</text>
1982
2059
  <circle cx="140.0" cy="99.0" r="2.3" fill="#10B981" class="lf-pop" style="--motion-delay:0ms"><title>完成时间 — positive evidence</title></circle>
1983
2060
  <circle cx="148.0" cy="99.0" r="2.3" fill="#10B981" class="lf-pop" style="--motion-delay:12ms"><title>完成时间 — positive evidence</title></circle>
@@ -2006,7 +2083,90 @@ select:focus-visible,
2006
2083
  <text x="128" y="256.0" text-anchor="end" font-size="9" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:740ms">过度依赖</text>
2007
2084
  <circle cx="140.0" cy="253.0" r="2.3" fill="#38BDF8" class="lf-pop" style="--motion-delay:700ms"><title>过度依赖 — negative evidence</title></circle>
2008
2085
  <text x="512" y="256.0" text-anchor="end" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-fade" style="--motion-delay:1200ms">-1</text>
2009
- </svg></div><figcaption class="lieflat-caption">基于 effect_direction 计数,全部数值来自 result.json。</figcaption><p class="lieflat-src">F5 Tick Rows · outcomes.direction_counts</p></figure><div class="lieflat-suppressed" role="note"><strong>已抑制 2 张图(数据不足,镜像 Meaningful Visualization Gate)</strong><ul><li><code>FOREST-PLOT (publication figure)</code>:fewer than 3 studies with numeric effect size (got 0)</li><li><code>L2 Dot Cascade</code>:fewer than 3 studies with numeric effect size (got 0)</li></ul></div></div></div></section><section class="brief-block brief-sources" id="brief-sources-zh"><header class="brief-block-header"><h2>关键来源</h2><p>摘要页只列最关键的来源;完整溯源在完整报告中展开。</p></header><div class="brief-block-body"><div class="brief-source-grid"><article class="brief-source"><code>S-2023-kazemitabaar</code><h3><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</a></h3><p>T1 DOI 可验证论文 · 2023</p></article><article class="brief-source"><code>S-2025-bastani</code><h3><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">Generative AI without guardrails can harm learning: Evidence from high school mathematics</a></h3><p>T1 DOI 可验证论文 · 2025</p></article><article class="brief-source"><code>S-2024-marzuki</code><h3><a href="https://link.springer.com/article/10.1186/s40561-024-00295-9">Impact of ChatGPT on ESL students&#x27; academic writing skills</a></h3><p>T1 DOI 可验证论文 · 2024</p></article><article class="brief-source"><code>S-2023-peng</code><h3><a href="https://doi.org/10.48550/arXiv.2302.06590">The Impact of AI on Developer Productivity: Evidence from GitHub Copilot</a></h3><p>tier2_academic_database · 2023</p></article></div><p class="brief-source-more">完整报告中还有 4 个来源可展开追溯。</p></div></section></div></div>
2086
+ </svg></div><figcaption class="lieflat-caption">基于 effect_direction 计数,全部数值来自 result.json。</figcaption><p class="lieflat-src">F5 Tick Rows · Outcomes.direction Counts</p></figure><figure class="lieflat-card" data-lieflat data-visual="lieflat-paired_rungs" data-chart-id="lieflat-paired-rungs.svg"><h3 class="lieflat-title">各结果类型的正负证据对照</h3><p class="lieflat-sub">左右两列分别汇总正向与负向证据条数</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 541 300" width="100%" height="100%" role="img" aria-label="各结果类型的正负证据对照" style="background:#111827;">
2087
+ <text x="60" y="76" font-size="8" font-weight="700" fill="#10B981" class="lf-fade" style="--motion-delay:40ms">正向</text>
2088
+ <text x="60" y="92" font-size="8" font-weight="700" fill="#38BDF8" class="lf-fade" style="--motion-delay:80ms">负向</text>
2089
+ <rect x="73.0" y="230.3" width="13" height="5.5" rx="2.6" fill="#10B981" class="lf-fade" style="--motion-delay:60ms"><title>知识获得 — positive</title></rect>
2090
+ <text x="90.0" y="254" text-anchor="middle" font-size="8.0" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:100ms">知识获得</text>
2091
+ <text x="90.0" y="222.3" text-anchor="middle" font-size="8" font-weight="800" fill="#F8FAFC" class="lf-fade" style="--motion-delay:520ms">1 / 0</text>
2092
+ <text x="150.0" y="254" text-anchor="middle" font-size="8.0" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:200ms">记忆保持</text>
2093
+ <text x="150.0" y="230.0" text-anchor="middle" font-size="8" font-weight="800" fill="#F8FAFC" class="lf-fade" style="--motion-delay:620ms">0 / 0</text>
2094
+ <rect x="214.0" y="230.3" width="13" height="5.5" rx="2.6" fill="#38BDF8" class="lf-fade" style="--motion-delay:260ms"><title>独立问题解决 — negative</title></rect>
2095
+ <text x="210.0" y="254" text-anchor="middle" font-size="8.0" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:300ms">独立问题解决</text>
2096
+ <text x="210.0" y="222.3" text-anchor="middle" font-size="8" font-weight="800" fill="#F8FAFC" class="lf-fade" style="--motion-delay:720ms">0 / 1</text>
2097
+ <rect x="253.0" y="230.3" width="13" height="5.5" rx="2.6" fill="#10B981" class="lf-fade" style="--motion-delay:360ms"><title>完成时间 — positive</title></rect>
2098
+ <rect x="253.0" y="222.6" width="13" height="5.5" rx="2.6" fill="#10B981" class="lf-fade" style="--motion-delay:372ms"><title>完成时间 — positive</title></rect>
2099
+ <text x="270.0" y="254" text-anchor="middle" font-size="8.0" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:400ms">完成时间</text>
2100
+ <text x="270.0" y="214.6" text-anchor="middle" font-size="8" font-weight="800" fill="#F8FAFC" class="lf-fade" style="--motion-delay:820ms">2 / 0</text>
2101
+ <text x="330.0" y="254" text-anchor="middle" font-size="8.0" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:500ms">代码质量</text>
2102
+ <text x="330.0" y="230.0" text-anchor="middle" font-size="8" font-weight="800" fill="#F8FAFC" class="lf-fade" style="--motion-delay:920ms">0 / 0</text>
2103
+ <rect x="373.0" y="230.3" width="13" height="5.5" rx="2.6" fill="#10B981" class="lf-fade" style="--motion-delay:560ms"><title>作业成绩 — positive</title></rect>
2104
+ <rect x="373.0" y="222.6" width="13" height="5.5" rx="2.6" fill="#10B981" class="lf-fade" style="--motion-delay:572ms"><title>作业成绩 — positive</title></rect>
2105
+ <text x="390.0" y="254" text-anchor="middle" font-size="8.0" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:600ms">作业成绩</text>
2106
+ <text x="390.0" y="214.6" text-anchor="middle" font-size="8" font-weight="800" fill="#F8FAFC" class="lf-fade" style="--motion-delay:1020ms">2 / 0</text>
2107
+ <rect x="433.0" y="230.3" width="13" height="5.5" rx="2.6" fill="#10B981" class="lf-fade" style="--motion-delay:660ms"><title>元认知 — positive</title></rect>
2108
+ <text x="450.0" y="254" text-anchor="middle" font-size="8.0" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:700ms">元认知</text>
2109
+ <text x="450.0" y="222.3" text-anchor="middle" font-size="8" font-weight="800" fill="#F8FAFC" class="lf-fade" style="--motion-delay:1120ms">1 / 0</text>
2110
+ <rect x="514.0" y="230.3" width="13" height="5.5" rx="2.6" fill="#38BDF8" class="lf-fade" style="--motion-delay:760ms"><title>过度依赖 — negative</title></rect>
2111
+ <text x="510.0" y="254" text-anchor="middle" font-size="8.0" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:800ms">过度依赖</text>
2112
+ <text x="510.0" y="222.3" text-anchor="middle" font-size="8" font-weight="800" fill="#F8FAFC" class="lf-fade" style="--motion-delay:1220ms">0 / 1</text>
2113
+ <line x1="46" y1="238" x2="512" y2="238" stroke="#F8FAFC" stroke-width="1.2" class="lf-draw" style="--motion-delay:120ms"/>
2114
+ </svg></div><figcaption class="lieflat-caption">当同一结果同时存在正向与负向证据时,分列呈现避免相互抵消。</figcaption><p class="lieflat-src">F6 Paired Rungs · Outcomes.paired Counts</p></figure><figure class="lieflat-card" data-lieflat data-visual="lieflat-brand_spectrum" data-chart-id="lieflat-brand-spectrum.svg"><h3 class="lieflat-title">各结果类型的净效应倾向</h3><p class="lieflat-sub">位置 =(正向 − 负向)÷ 方向计数 · 中点为中性</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 540 360" width="100%" height="100%" role="img" aria-label="各结果类型的净效应倾向" style="background:#111827;">
2115
+ <text x="138" y="62" text-anchor="end" font-size="9" font-weight="700" fill="#94A3B8" class="lf-fade" style="--motion-delay:40ms">负向主导</text>
2116
+ <text x="412" y="62" font-size="9" font-weight="700" fill="#94A3B8" class="lf-fade" style="--motion-delay:80ms">正向主导</text>
2117
+ <path d="M 400.0 88 C 400.0 111.0, 150.0 111.0, 150.0 134 C 150.0 157.0, 400.0 157.0, 400.0 180 C 400.0 203.0, 400.0 203.0, 400.0 226 C 400.0 249.0, 400.0 249.0, 400.0 272 C 400.0 295.0, 150.0 295.0, 150.0 318" fill="none" stroke="#1E293B" stroke-width="26" stroke-linecap="round" stroke-linejoin="round" opacity="0.95" class="lf-draw" style="--motion-delay:60ms"/>
2118
+ <line x1="150" y1="88" x2="400" y2="88" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:60ms"/>
2119
+ <line x1="150" y1="84" x2="150" y2="92" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:60ms"/>
2120
+ <line x1="400" y1="84" x2="400" y2="92" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:60ms"/>
2121
+ <text x="134" y="91" text-anchor="end" font-size="8.5" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:90ms">知识获得</text>
2122
+ <circle cx="400.0" cy="88" r="7.5" fill="#10B981" stroke="#111827" stroke-width="1.8" class="lf-pop" style="--motion-delay:160ms"><title>知识获得 — 负向主导↔正向主导: +100% (pos 1 / neg 0 / null 0)</title></circle>
2123
+ <text x="400.0" y="77" fill="#F8FAFC" font-size="8" font-weight="800" text-anchor="middle" class="lf-fade" style="--motion-delay:220ms">+100%</text>
2124
+ <line x1="150" y1="134" x2="400" y2="134" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:160ms"/>
2125
+ <line x1="150" y1="130" x2="150" y2="138" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:160ms"/>
2126
+ <line x1="400" y1="130" x2="400" y2="138" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:160ms"/>
2127
+ <text x="134" y="137" text-anchor="end" font-size="8.5" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:190ms">独立问题解决</text>
2128
+ <circle cx="150.0" cy="134" r="7.5" fill="#38BDF8" stroke="#111827" stroke-width="1.8" class="lf-pop" style="--motion-delay:260ms"><title>独立问题解决 — 负向主导↔正向主导: -100% (pos 0 / neg 1 / null 2)</title></circle>
2129
+ <text x="150.0" y="123" fill="#F8FAFC" font-size="8" font-weight="800" text-anchor="middle" class="lf-fade" style="--motion-delay:320ms">-100%</text>
2130
+ <line x1="150" y1="180" x2="400" y2="180" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:260ms"/>
2131
+ <line x1="150" y1="176" x2="150" y2="184" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:260ms"/>
2132
+ <line x1="400" y1="176" x2="400" y2="184" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:260ms"/>
2133
+ <text x="134" y="183" text-anchor="end" font-size="8.5" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:290ms">完成时间</text>
2134
+ <circle cx="400.0" cy="180" r="7.5" fill="#10B981" stroke="#111827" stroke-width="1.8" class="lf-pop" style="--motion-delay:360ms"><title>完成时间 — 负向主导↔正向主导: +100% (pos 2 / neg 0 / null 0)</title></circle>
2135
+ <text x="400.0" y="169" fill="#F8FAFC" font-size="8" font-weight="800" text-anchor="middle" class="lf-fade" style="--motion-delay:420ms">+100%</text>
2136
+ <line x1="150" y1="226" x2="400" y2="226" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:360ms"/>
2137
+ <line x1="150" y1="222" x2="150" y2="230" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:360ms"/>
2138
+ <line x1="400" y1="222" x2="400" y2="230" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:360ms"/>
2139
+ <text x="134" y="229" text-anchor="end" font-size="8.5" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:390ms">作业成绩</text>
2140
+ <circle cx="400.0" cy="226" r="7.5" fill="#10B981" stroke="#111827" stroke-width="1.8" class="lf-pop" style="--motion-delay:460ms"><title>作业成绩 — 负向主导↔正向主导: +100% (pos 2 / neg 0 / null 0)</title></circle>
2141
+ <text x="400.0" y="215" fill="#F8FAFC" font-size="8" font-weight="800" text-anchor="middle" class="lf-fade" style="--motion-delay:520ms">+100%</text>
2142
+ <line x1="150" y1="272" x2="400" y2="272" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:460ms"/>
2143
+ <line x1="150" y1="268" x2="150" y2="276" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:460ms"/>
2144
+ <line x1="400" y1="268" x2="400" y2="276" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:460ms"/>
2145
+ <text x="134" y="275" text-anchor="end" font-size="8.5" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:490ms">元认知</text>
2146
+ <circle cx="400.0" cy="272" r="7.5" fill="#10B981" stroke="#111827" stroke-width="1.8" class="lf-pop" style="--motion-delay:560ms"><title>元认知 — 负向主导↔正向主导: +100% (pos 1 / neg 0 / null 0)</title></circle>
2147
+ <text x="400.0" y="261" fill="#F8FAFC" font-size="8" font-weight="800" text-anchor="middle" class="lf-fade" style="--motion-delay:620ms">+100%</text>
2148
+ <line x1="150" y1="318" x2="400" y2="318" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:560ms"/>
2149
+ <line x1="150" y1="314" x2="150" y2="322" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:560ms"/>
2150
+ <line x1="400" y1="314" x2="400" y2="322" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:560ms"/>
2151
+ <text x="134" y="321" text-anchor="end" font-size="8.5" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:590ms">过度依赖</text>
2152
+ <circle cx="150.0" cy="318" r="7.5" fill="#38BDF8" stroke="#111827" stroke-width="1.8" class="lf-pop" style="--motion-delay:660ms"><title>过度依赖 — 负向主导↔正向主导: -100% (pos 0 / neg 1 / null 0)</title></circle>
2153
+ <text x="150.0" y="307" fill="#F8FAFC" font-size="8" font-weight="800" text-anchor="middle" class="lf-fade" style="--motion-delay:720ms">-100%</text>
2154
+ <text x="30" y="338" font-size="8" font-weight="600" fill="#64748B" class="lf-fade" style="--motion-delay:400ms">position = (positive − negative) ÷ total direction counts</text>
2155
+ </svg></div><figcaption class="lieflat-caption">双极展示各结果构念整体偏向支持还是反对。</figcaption><p class="lieflat-src">L7 Brand Spectrum · Outcomes.bipolar Axes</p></figure><figure class="lieflat-card" data-lieflat data-visual="lieflat-hundred_field" data-chart-id="lieflat-hundred-field.svg"><h3 class="lieflat-title">研究设计构成</h3><p class="lieflat-sub">每格 = 1 篇研究 · 显示证据来自哪些研究设计</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 540 300" width="100%" height="100%" role="img" aria-label="研究设计构成" style="background:#111827;">
2156
+ <rect x="40.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#38BDF8" class="lf-pop" style="--motion-delay:0ms"><title>rct — 1 study</title></rect>
2157
+ <rect x="58.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#38BDF8" class="lf-pop" style="--motion-delay:12ms"><title>rct — 1 study</title></rect>
2158
+ <rect x="76.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#38BDF8" class="lf-pop" style="--motion-delay:24ms"><title>rct — 1 study</title></rect>
2159
+ <rect x="94.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#38BDF8" class="lf-pop" style="--motion-delay:36ms"><title>rct — 1 study</title></rect>
2160
+ <rect x="112.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#38BDF8" class="lf-pop" style="--motion-delay:48ms"><title>rct — 1 study</title></rect>
2161
+ <rect x="130.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#38BDF8" class="lf-pop" style="--motion-delay:60ms"><title>rct — 1 study</title></rect>
2162
+ <rect x="148.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#38BDF8" class="lf-pop" style="--motion-delay:72ms"><title>rct — 1 study</title></rect>
2163
+ <rect x="166.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#10B981" class="lf-pop" style="--motion-delay:84ms"><title>observational — 1 study</title></rect>
2164
+ <rect x="184.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#10B981" class="lf-pop" style="--motion-delay:96ms"><title>observational — 1 study</title></rect>
2165
+ <rect x="202.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#10B981" class="lf-pop" style="--motion-delay:108ms"><title>observational — 1 study</title></rect>
2166
+ <rect x="40.0" y="92.0" width="15.5" height="15.5" rx="3" fill="#F59E0B" class="lf-pop" style="--motion-delay:120ms"><title>mixed_methods — 1 study</title></rect>
2167
+ <rect x="58.0" y="92.0" width="15.5" height="15.5" rx="3" fill="#64748B" class="lf-pop" style="--motion-delay:132ms"><title>qualitative — 1 study</title></rect>
2168
+ <text x="30" y="270" font-size="8" font-weight="600" fill="#64748B" class="lf-fade" style="--motion-delay:400ms">每格 = 1 篇研究</text>
2169
+ </svg></div><figcaption class="lieflat-caption">当证据包含多种研究设计时,构成图比表格更快暴露设计偏斜。</figcaption><p class="lieflat-src">L14 Hundred Field · Evidence.study Type Composition</p></figure></div></div></section><section class="brief-block brief-sources" id="brief-sources-zh"><header class="brief-block-header"><h2>关键来源</h2><p>摘要页只列最关键的来源;完整溯源在完整报告中展开。</p></header><div class="brief-block-body"><div class="brief-source-grid"><article class="brief-source"><code>S-2023-kazemitabaar</code><h3><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</a></h3><p>T1 DOI 可验证论文 · 2023</p></article><article class="brief-source"><code>S-2025-bastani</code><h3><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">Generative AI without guardrails can harm learning: Evidence from high school mathematics</a></h3><p>T1 DOI 可验证论文 · 2025</p></article><article class="brief-source"><code>S-2024-marzuki</code><h3><a href="https://link.springer.com/article/10.1186/s40561-024-00295-9">Impact of ChatGPT on ESL students&#x27; academic writing skills</a></h3><p>T1 DOI 可验证论文 · 2024</p></article><article class="brief-source"><code>S-2023-peng</code><h3><a href="https://doi.org/10.48550/arXiv.2302.06590">The Impact of AI on Developer Productivity: Evidence from GitHub Copilot</a></h3><p>Tier2 Academic Database · 2023</p></article></div><p class="brief-source-more">完整报告中还有 4 个来源可展开追溯。</p></div></section></div></div>
2010
2170
  <div class="report-page report-page-full" data-report-page="full" hidden>
2011
2171
  <div class="full-report-intro"><h2>完整报告</h2><p>结论前置:全部可追溯证据与方法学细节都在这里,关键论证位置穿插有意义的可视化,每个数字都能回查到 result.json。</p></div>
2012
2172
  <div class="full-report-layout"><aside class="full-report-toc" aria-label="目录"><div class="toc-head"><strong>目录</strong><button type="button" class="toc-collapse" aria-expanded="true" data-label-collapse="收起目录" data-label-expand="展开目录">收起目录</button></div><nav><a href="#full-01-decision" data-toc-target="full-01-decision" data-chapter-key="decision">01 结论、裁决与研究边界</a><a href="#full-02-evidence" data-toc-target="full-02-evidence" data-chapter-key="evidence">02 关键证据与结果分离</a><a href="#full-03-quality" data-toc-target="full-03-quality" data-chapter-key="quality">03 证据可信度、反证与方法审计</a><a href="#full-04-action" data-toc-target="full-04-action" data-chapter-key="action">04 适用范围与教学行动</a><a href="#full-05-evaluation" data-toc-target="full-05-evaluation" data-chapter-key="evaluation">05 试点设计、评估与停止条件</a><a href="#full-06-sources" data-toc-target="full-06-sources" data-chapter-key="sources">06 来源、溯源与附录</a></nav></aside><main class="full-report-content"><section id="full-01-decision" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>01 结论、裁决与研究边界</h2><p class="full-chapter-lead">先明确最终裁决与研究边界,再解释为什么。</p></header><div class="full-chapter-body">
@@ -2018,13 +2178,13 @@ select:focus-visible,
2018
2178
  </div>
2019
2179
  <p class="hero-rationale">任务表现的正面证据 + 有据可查的无护栏风险 + 混合的质量/可用性信号 + 大学层面学习证据缺失 → 有界、护栏化、带评估的试点,而非全面采用。</p>
2020
2180
  <div class="hero-insights">
2021
- <article class="hero-insight support"><span>最强支持结论</span><p class="hero-insight-text">AI 编程助手在训练期提升新手任务表现。</p></article>
2022
- <article class="hero-insight uncertain"><span>关键不确定性 / 反例</span><p class="hero-insight-text">AI 编程助手能否真正改善或保持大学新手的编程学习——本证据集中没有大学层面的直接 RCT [无直接证据]</p></article>
2023
- <article class="hero-insight risk"><span>主要风险</span><p class="hero-insight-text">无护栏使用的 AI 依赖与过度依赖风险真实且有记录(E-004),可用性发现亦予印证(E-012)。</p></article>
2024
- <article class="hero-insight next"><span>下一步</span><p class="hero-insight-text">当前结果未提供此项信息。</p></article>
2181
+ <article class="hero-insight support"><span>最强支持结论</span><p class="hero-insight-text">AI 编程助手在训练期稳定提升练习效率:69 名新手的随机对照中完成率 1.15 倍、用时 0.57 倍。</p></article>
2182
+ <article class="hero-insight uncertain"><span>关键不确定性 / 反例</span><p class="hero-insight-text">缺少大学层面的直接学习证据;唯一大规模试验显示,无护栏使用 GPT-4 的学生独立考试成绩下降 17%。</p></article>
2183
+ <article class="hero-insight risk"><span>主要风险</span><p class="hero-insight-text">无护栏使用会抬高练习表现却压低独立考试表现,而学习者往往意识不到这一落差。</p></article>
2184
+ <article class="hero-insight next"><span>下一步</span><p class="hero-insight-text">开展分阶段 CS1 试点:给提示而非答案、每周实验课使用、并以无 AI 迁移考试作为可叫停的验收条件。</p></article>
2025
2185
  </div>
2026
2186
  <p class="hero-provenance"><span>证据 / 来源</span> · 12 / 8</p>
2027
- </div><div class="scope-grid"><article class="scope-card"><h3>研究问题</h3><p class="scope-text">我们准备在大学一年级 C 语言课程中允许学生使用生成式 AI 编程助手。它到底会不会提高学习效果?应该怎样引入?</p></article><article class="scope-card"><h3>目标学习者</h3><p class="scope-text">教育阶段:大学一年级;专业:计算机科学与技术;先验知识:first_programming_course_no_prior_text_based_programming;学习者特征:mixed_ability_large_class_60_students</p></article><article class="scope-card"><h3>课程情境</h3><p class="scope-text">课程:C 语言程序设计;课程类型:lecture_lab;课程周期:16_weeks_one_semester</p></article><article class="scope-card"><h3>AI 干预</h3><p class="scope-text">teaching_method:lecture_with_lab_exercises;AI 工具:generative_ai_coding_assistant;允许使用:under_design_pending_evidence_review;使用频率:weekly_lab_sessions;干预周期:one_semester</p></article><article class="scope-card"><h3>比较条件</h3><p class="scope-text">no_ai_coding_assistant_control</p></article><article class="scope-card"><h3>结果构念</h3><p class="scope-text">主要结果:独立问题解决、代码质量;次要结果:完成时间、记忆保持、知识获得;风险结果:AI 依赖、过度依赖、迁移下降</p></article><article class="scope-card"><h3>研究范围</h3><p class="scope-text">时间范围:2021-2026;地域:worldwide;研究设计:随机对照试验、准实验、观察性研究</p></article><article class="scope-card"><h3>决策成功条件</h3><p class="scope-text">independent problem solving and code quality improve (or do not decline) while AI dependency risk stays controlled; evidence base supports a bounded pilot.</p></article></div></div></section><section id="full-02-evidence" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>02 关键证据与结果分离</h2><p class="full-chapter-lead">把任务表现、真实学习、保持与风险放在同一证据地图中,但不混为一谈。</p></header><div class="full-chapter-body"><div class="retrieval-grid"><article><h3>纳入标准</h3><ul><li>studies_of_generative_AI_coding_tools_in_learning_to_program</li><li>outcomes_measuring_learning_not_only_task_speed</li><li>university_or_novice_programming_populations</li></ul></article><article><h3>排除标准</h3><ul><li>practitioner_anecdotes_without_data</li><li>industry_professional_populations_only</li></ul></article></div><div class="retrieval-coverage"><h3>证据来源覆盖</h3><p><code>S-2023-kazemitabaar</code> <code>S-2025-bastani</code> <code>S-2024-marzuki</code> <code>S-2023-peng</code> <code>S-2023-yetistiren</code> <code>S-2022-finnie-ansley</code> <code>S-2023-explanations-compare</code> <code>S-2022-vaithilingam</code></p><p class="retrieval-note">当前报告只展示 result 中真实存在的检索与来源信息;没有流程计数时不伪造 PRISMA / funnel 数字。</p></div><div class="outcome-separation" data-visual="outcome-separation"><div class="visual-heading"><h3>结果分离 · 任务表现 ≠ 学习效果</h3><p>将不同结果类型分开裁决,避免把训练时更快、更高分直接等同于真正学会。</p><p class="semantic-note">效应方向来自 evidence.effect_direction;“支持某个主张”不等于“结果是正向”。</p></div><div class="outcome-groups"><article class="outcome-group outcome-task"><h3>任务 / 近端表现</h3><ul><li><strong>完成时间</strong><span class="outcome-states"><span class="dir pos">正向效应 2</span></span></li><li><strong>代码质量</strong><span class="outcome-states"><span class="dir neu">零效应 1</span></span></li><li><strong>作业成绩</strong><span class="outcome-states"><span class="dir pos">正向效应 2</span></span></li></ul></article><article class="outcome-group outcome-learning"><h3>学习 / 保持 / 迁移</h3><ul><li><strong>知识获得</strong><span class="outcome-states"><span class="dir pos">正向效应 1</span></span></li><li><strong>记忆保持</strong><span class="outcome-states"><span class="dir neu">零效应 1</span></span></li><li><strong>独立问题解决</strong><span class="outcome-states"><span class="dir neg">负向效应 1</span><span class="dir neu">零效应 2</span></span></li></ul></article><article class="outcome-group outcome-risk"><h3>风险 / 依赖</h3><ul><li><strong>过度依赖</strong><span class="outcome-states"><span class="dir neg">负向效应 1</span></span></li></ul></article><article class="outcome-group outcome-other"><h3>其他结果</h3><ul><li><strong>元认知</strong><span class="outcome-states"><span class="dir pos">正向效应 1</span></span></li></ul></article></div></div><div class='table-wrap outcome-table'><table class='data-table'><thead><tr><th>结果类型</th><th>正向效应</th><th>负向效应</th><th>零效应</th><th>证据</th></tr></thead><tbody><tr><td><strong>知识获得</strong><span class='raw-tag' title='原始标识'>knowledge_gain</span></td><td class='num'>1</td><td class='num'>0</td><td class='num'>0</td><td><code>E-007</code> </td></tr><tr><td><strong>记忆保持</strong><span class='raw-tag' title='原始标识'>retention</span></td><td class='num'>0</td><td class='num'>0</td><td class='num'>1</td><td><code>E-003</code> </td></tr><tr><td><strong>独立问题解决</strong><span class='raw-tag' title='原始标识'>independent_problem_solving</span></td><td class='num'>0</td><td class='num'>1</td><td class='num'>2</td><td><code>E-002</code> <code>E-004</code> <code>E-005</code> </td></tr><tr><td><strong>完成时间</strong><span class='raw-tag' title='原始标识'>completion_time</span></td><td class='num'>2</td><td class='num'>0</td><td class='num'>0</td><td><code>E-001</code> <code>E-008</code> </td></tr><tr><td><strong>代码质量</strong><span class='raw-tag' title='原始标识'>code_quality</span></td><td class='num'>0</td><td class='num'>0</td><td class='num'>1</td><td><code>E-009</code> </td></tr><tr><td><strong>作业成绩</strong><span class='raw-tag' title='原始标识'>assignment_score</span></td><td class='num'>2</td><td class='num'>0</td><td class='num'>0</td><td><code>E-006</code> <code>E-010</code> </td></tr><tr><td><strong>元认知</strong><span class='raw-tag' title='原始标识'>metacognition</span></td><td class='num'>1</td><td class='num'>0</td><td class='num'>0</td><td><code>E-011</code> </td></tr><tr><td><strong>过度依赖</strong><span class='raw-tag' title='原始标识'>over_reliance</span></td><td class='num'>0</td><td class='num'>1</td><td class='num'>0</td><td><code>E-012</code> </td></tr></tbody></table></div><div class="visual-surface" data-visual="outcome-evidence-balance"><svg viewBox="0 0 720 300" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="各结果类型证据效应分布"><title>各结果类型证据效应分布</title><desc>各结果类型的正向 / 负向 / 零效应证据条数(基于 effect_direction,不等同于 Claim 是否被支持)。</desc><rect x="0" y="0" width="720" height="300" fill="#FFFFFF"/><line x1="415.0" y1="46" x2="415.0" y2="182" stroke="#999" stroke-width="1" stroke-dasharray="3,3"/><text x="142" y="58.0" text-anchor="end" font-size="11" fill="#333">知识获得</text><rect x="415.0" y="49.8" width="132.5" height="9.4" fill="#5E8A6A"/><text x="481.2" y="58.0" text-anchor="middle" font-size="9" fill="#fff">1</text><text x="142" y="75.0" text-anchor="end" font-size="11" fill="#333">记忆保持</text><text x="142" y="92.0" text-anchor="end" font-size="11" fill="#333">独立问题解决</text><rect x="282.5" y="83.8" width="132.5" height="9.4" fill="#A85B53"/><text x="348.8" y="92.0" text-anchor="middle" font-size="9" fill="#fff">-1</text><text x="142" y="109.0" text-anchor="end" font-size="11" fill="#333">完成时间</text><rect x="415.0" y="100.8" width="265.0" height="9.4" fill="#5E8A6A"/><text x="547.5" y="109.0" text-anchor="middle" font-size="9" fill="#fff">2</text><text x="142" y="126.0" text-anchor="end" font-size="11" fill="#333">代码质量</text><text x="142" y="143.0" text-anchor="end" font-size="11" fill="#333">作业成绩</text><rect x="415.0" y="134.8" width="265.0" height="9.4" fill="#5E8A6A"/><text x="547.5" y="143.0" text-anchor="middle" font-size="9" fill="#fff">2</text><text x="142" y="160.0" text-anchor="end" font-size="11" fill="#333">元认知</text><rect x="415.0" y="151.8" width="132.5" height="9.4" fill="#5E8A6A"/><text x="481.2" y="160.0" text-anchor="middle" font-size="9" fill="#fff">1</text><text x="142" y="177.0" text-anchor="end" font-size="11" fill="#333">过度依赖</text><rect x="282.5" y="168.8" width="132.5" height="9.4" fill="#A85B53"/><text x="348.8" y="177.0" text-anchor="middle" font-size="9" fill="#fff">-1</text><rect x="415.0" y="206.1" width="8.0" height="6" fill="#C99A4A"/><rect x="415.0" y="214.9" width="8.0" height="6" fill="#C99A4A"/><rect x="415.0" y="232.4" width="8.0" height="6" fill="#C99A4A"/><rect x="150" y="14" width="10" height="10" fill="#5E8A6A"/><text x="164" y="23" font-size="10" fill="#333">正向效应</text><rect x="226" y="14" width="10" height="10" fill="#A85B53"/><text x="240" y="23" font-size="10" fill="#333">负向效应</text><rect x="302" y="14" width="10" height="10" fill="#C99A4A"/><text x="316" y="23" font-size="10" fill="#333">零效应</text></svg><p class="chart-interpretation"><strong>这意味着什么:</strong>各 Outcome 的正向 / 负向 / 零效应证据条数对比。这里展示的是 effect_direction,不是“这条证据是否支持某个主张”。</p></div><div id="chart-outcome-zh" class="chart-mount" aria-label="结果证据概览"></div><figure class="academic-figure" data-visual="outcome-evidence-balance"><svg viewBox="0 0 720 300" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="各结果类型效应方向分布(出版级学术图)"><title>各结果类型效应方向分布(出版级学术图)</title><desc>各结果类型的正向 / 负向 / 零效应证据条数;计数轴整数刻度,不随主题变化。来源:EduEvidence result.json。</desc><rect width="720" height="300" fill="#FFFFFF"/><line x1="70" y1="250" x2="650" y2="250" stroke="#333" stroke-width="1"/><text x="106.2" y="266" text-anchor="middle" font-size="10" fill="#333">知识获得</text><rect x="88.1" y="150.0" width="18.1" height="100.0" fill="#38BDF8"/><text x="97.2" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><rect x="106.2" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="124.4" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><text x="178.8" y="266" text-anchor="middle" font-size="10" fill="#333">记忆保持</text><rect x="160.6" y="250.0" width="18.1" height="1.0" fill="#38BDF8"/><rect x="178.8" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="196.9" y="150.0" width="18.1" height="100.0" fill="#F59E0B"/><text x="205.9" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><text x="251.2" y="266" text-anchor="middle" font-size="10" fill="#333">独立问题解决</text><rect x="233.1" y="250.0" width="18.1" height="1.0" fill="#38BDF8"/><rect x="251.2" y="150.0" width="18.1" height="100.0" fill="#10B981"/><text x="260.3" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><rect x="269.4" y="50.0" width="18.1" height="200.0" fill="#F59E0B"/><text x="278.4" y="47.0" text-anchor="middle" font-size="9" fill="#333">2</text><text x="323.8" y="266" text-anchor="middle" font-size="10" fill="#333">完成时间</text><rect x="305.6" y="50.0" width="18.1" height="200.0" fill="#38BDF8"/><text x="314.7" y="47.0" text-anchor="middle" font-size="9" fill="#333">2</text><rect x="323.8" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="341.9" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><text x="396.2" y="266" text-anchor="middle" font-size="10" fill="#333">代码质量</text><rect x="378.1" y="250.0" width="18.1" height="1.0" fill="#38BDF8"/><rect x="396.2" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="414.4" y="150.0" width="18.1" height="100.0" fill="#F59E0B"/><text x="423.4" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><text x="468.8" y="266" text-anchor="middle" font-size="10" fill="#333">作业成绩</text><rect x="450.6" y="50.0" width="18.1" height="200.0" fill="#38BDF8"/><text x="459.7" y="47.0" text-anchor="middle" font-size="9" fill="#333">2</text><rect x="468.8" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="486.9" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><text x="541.2" y="266" text-anchor="middle" font-size="10" fill="#333">元认知</text><rect x="523.1" y="150.0" width="18.1" height="100.0" fill="#38BDF8"/><text x="532.2" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><rect x="541.2" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="559.4" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><text x="613.8" y="266" text-anchor="middle" font-size="10" fill="#333">过度依赖</text><rect x="595.6" y="250.0" width="18.1" height="1.0" fill="#38BDF8"/><rect x="613.8" y="150.0" width="18.1" height="100.0" fill="#10B981"/><text x="622.8" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><rect x="631.9" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><line x1="65" y1="250.0" x2="70" y2="250.0" stroke="#999"/><text x="62" y="253.0" text-anchor="end" font-size="9" fill="#666">0</text><line x1="65" y1="150.0" x2="70" y2="150.0" stroke="#999"/><text x="62" y="153.0" text-anchor="end" font-size="9" fill="#666">1</text><line x1="65" y1="50.0" x2="70" y2="50.0" stroke="#999"/><text x="62" y="53.0" text-anchor="end" font-size="9" fill="#666">2</text><text x="360.0" y="30" text-anchor="middle" font-size="14" font-weight="700" fill="#111">各结果类型的效应方向分布</text><rect x="70" y="8" width="10" height="10" fill="#38BDF8"/><text x="84" y="17" font-size="10" fill="#333">正向效应</text><rect x="146" y="8" width="10" height="10" fill="#10B981"/><text x="160" y="17" font-size="10" fill="#333">负向效应</text><rect x="222" y="8" width="10" height="10" fill="#F59E0B"/><text x="236" y="17" font-size="10" fill="#333">零效应</text><text x="20" y="290" font-size="11" fill="#333333" font-style="italic">图 1. 各结果类型的正向 / 负向 / 零效应证据条数(基于 effect_direction;出版级学术图,不随主题变化)。来源:EduEvidence result.json。</text></svg><figcaption>图 1. 各结果类型的正向 / 负向 / 零效应证据数量(基于 effect_direction,不等同于 Claim 是否被支持)。</figcaption></figure><div class='matrix-controls'><div class='matrix-tools'><input id='matrix-search-full-zh' type='search' placeholder='搜索证据…' aria-label='筛选 / 搜索证据'><select id='matrix-direction-full-zh' aria-label='按效应方向筛选'><option value=''>全部效应</option><option value='positive'>正向效应</option><option value='negative'>负向效应</option><option value='null'>零效应</option></select><select id='matrix-outcome-full-zh' aria-label='按结果类型筛选'><option value=''>全部结果</option><option value='assignment_score'>作业成绩</option><option value='code_quality'>代码质量</option><option value='completion_time'>完成时间</option><option value='independent_problem_solving'>独立问题解决</option><option value='knowledge_gain'>知识获得</option><option value='metacognition'>元认知</option><option value='over_reliance'>过度依赖</option><option value='retention'>记忆保持</option></select></div></div><div class='table-wrap matrix-wrap'><table id='evidence-matrix-full-zh' class='data-table evidence-matrix'><thead><tr><th>ID</th><th>结果</th><th>效应</th><th>质量</th><th>主张</th><th>来源</th></tr></thead><tbody><tr data-effect="positive" data-direction="positive" data-outcome="completion_time" data-search="e-001 study-kazemitabaar-2023 smpl-kazemitabaar-2023-n69 studying the effect of ai code generators on supporting novice learners in introductory programming 在练习环节,无护栏的 gpt base(类标准 chatgpt 界面)使高中生的练习成绩相对对照组提高 48%,带护栏的 gpt tutor 提高 127%(table 1:practice 系数 0.137/0.361,对照均值 0.284) 土耳其高中 9-11 年级数学学生(约 1000 名学生、4 次 90 分钟课内环节,共 2848 观测) 三臂 rct:gpt base(无护栏标准 chatgpt 式界面)与 gpt tutor(护栏版,教师设计提示、不给直接答案)用于数学练习 对照组(无 ai 传统教学) positive s-2023-kazemitabaar"><td><code>E-001</code></td><td><strong>完成时间</strong></td><td><span class="dir pos">正向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">9</span><span class="quality-meter" aria-hidden="true"><i style="width:90%"></i></span></div></td><td class="claim-cell"><p>在练习环节,无护栏的 GPT Base(类标准 ChatGPT 界面)使高中生的练习成绩相对对照组提高 48%,带护栏的 GPT Tutor 提高 127%(Table 1:practice 系数 0.137/0.361,对照均值 0.284)</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-KAZEMITABAAR-2023</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-KAZEMITABAAR-2023-N69</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>随机对照试验</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>k12_ages_10_17</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>土耳其高中 9-11 年级数学学生(约 1000 名学生、4 次 90 分钟课内环节,共 2848 观测)</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>69</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>三臂 RCT:GPT Base(无护栏标准 ChatGPT 式界面)与 GPT Tutor(护栏版,教师设计提示、不给直接答案)用于数学练习</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>对照组(无 AI 传统教学)</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>code_authoring_task_progress_and_time</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>1.15x completion rate, 0.57x time, 1.8x correctness</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>正向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>3_weeks_training</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>controlled experiment with random assignment, immediate post-test and 1-week retention test</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>randomized_controlled_design;immediate_post_test_and_retention_test;code_modification_task_guard</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>non_university_population_ages_10_17;small_sample_69;self-paced environment differs from classroom</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>prior_programming_competency_interaction</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1_study_design=2 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=2 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>9.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=partial_novice_programmers_but_younger · subject_match=introductory_programming · tool_match=codex_like_generative_ai · scope=task_performance_during_training</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.7</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>在练习环节,无护栏的 GPT Base(类标准 ChatGPT 界面)使高中生的练习成绩相对对照组提高 48%,带护栏的 GPT Tutor 提高 127%(Table 1:practice 系数 0.137/0.361,对照均值 0.284)</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3544548.3580919</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">https://dl.acm.org/doi/10.1145/3544548.3580919</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3544548.3580919" title="Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming"><code>S-2023-kazemitabaar</code></a></td></tr><tr data-effect="null" data-direction="null" data-outcome="independent_problem_solving" data-search="e-002 study-kazemitabaar-2023 smpl-kazemitabaar-2023-n69 studying the effect of ai code generators on supporting novice learners in introductory programming 移除 ai 访问后的无辅助独立考试中,gpt base 组成绩比从未使用 ai 的对照组低 17%(统计显著),表明无护栏使用 ai 损害技能习得;机制上学生把 gpt 当&#x27;拐杖&#x27;直接抄答案(gpt base 答对率仅 51%,其中 42% 逻辑错误、8% 算术错误),且学生自评过度乐观、未察觉学习受损 土耳其高中 9-11 年级数学学生(约 1000 名学生,共 2848 观测) 无护栏 gpt base(类标准 chatgpt 界面)课内练习;移除访问后参加独立考试 对照组(从未使用 ai) null s-2023-kazemitabaar"><td><code>E-002</code></td><td><strong>独立问题解决</strong></td><td><span class="dir neu">零效应</span><span class="relation-note">与主张关系: 中性</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>移除 AI 访问后的无辅助独立考试中,GPT Base 组成绩比从未使用 AI 的对照组低 17%(统计显著),表明无护栏使用 AI 损害技能习得;机制上学生把 GPT 当&#x27;拐杖&#x27;直接抄答案(GPT Base 答对率仅 51%,其中 42% 逻辑错误、8% 算术错误),且学生自评过度乐观、未察觉学习受损</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-KAZEMITABAAR-2023</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-KAZEMITABAAR-2023-N69</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>随机对照试验</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>k12_ages_10_17</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>土耳其高中 9-11 年级数学学生(约 1000 名学生,共 2848 观测)</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>69</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>无护栏 GPT Base(类标准 ChatGPT 界面)课内练习;移除访问后参加独立考试</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>对照组(从未使用 AI)</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>manual code-modification tasks during training</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>no significant difference between groups</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>零效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>中性</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>3_weeks_training</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>controlled experiment, code-modification task followed each code-authoring task</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>direct_test_of_transfer-adjacent_skill;same_session_measurement</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>code modification is not full independent problem solving;non_university population</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>practice_effect</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1_study_design=2 · D2_sample_quality=2 · D3_measurement_validity=1 · D4_temporal_strength=1 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=partial · subject_match=introductory_programming · tool_match=codex · scope=short-term manual code modification</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.5</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>移除 AI 访问后的无辅助独立考试中,GPT Base 组成绩比从未使用 AI 的对照组低 17%(统计显著),表明无护栏使用 AI 损害技能习得;机制上学生把 GPT 当&#x27;拐杖&#x27;直接抄答案(GPT Base 答对率仅 51%,其中 42% 逻辑错误、8% 算术错误),且学生自评过度乐观、未察觉学习受损</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3544548.3580919</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">https://dl.acm.org/doi/10.1145/3544548.3580919</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3544548.3580919" title="Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming"><code>S-2023-kazemitabaar</code></a></td></tr><tr data-effect="null" data-direction="null" data-outcome="retention" data-search="e-003 study-kazemitabaar-2023 smpl-kazemitabaar-2023-n69 studying the effect of ai code generators on supporting novice learners in introductory programming 带护栏的 gpt tutor(教师设计提示而非直接答案)在练习成绩 +127% 的同时,移除访问后的独立考试负效应基本消除(-0.004,不显著),说明精心设计的护栏可兼得练习提升与学习保持 土耳其高中 9-11 年级数学学生(约 1000 名学生,共 2848 观测) gpt tutor(护栏版:教师设计提示、不给直接答案)用于数学练习 对照组(无 ai)与 gpt base(无护栏)组 null s-2023-kazemitabaar"><td><code>E-003</code></td><td><strong>记忆保持</strong></td><td><span class="dir neu">零效应</span><span class="relation-note">与主张关系: 中性</span></td><td><div class="quality-cell"><span class="num">9</span><span class="quality-meter" aria-hidden="true"><i style="width:90%"></i></span></div></td><td class="claim-cell"><p>带护栏的 GPT Tutor(教师设计提示而非直接答案)在练习成绩 +127% 的同时,移除访问后的独立考试负效应基本消除(-0.004,不显著),说明精心设计的护栏可兼得练习提升与学习保持</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-KAZEMITABAAR-2023</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-KAZEMITABAAR-2023-N69</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>随机对照试验</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>k12_ages_10_17</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>土耳其高中 9-11 年级数学学生(约 1000 名学生,共 2848 观测)</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>69</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>GPT Tutor(护栏版:教师设计提示、不给直接答案)用于数学练习</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>对照组(无 AI)与 GPT Base(无护栏)组</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>retention post-test one week after training</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>slightly better for Codex group but not significant</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>零效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>中性</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>3_weeks_training_plus_1_week_retention</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>controlled experiment with delayed retention test</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>delayed_test_included</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>1-week retention window is short;small sample;non-university population</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>prior_competency</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1_study_design=2 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=2 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>9.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=partial · subject_match=introductory_programming · tool_match=codex · scope=retention over one week</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.5</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>带护栏的 GPT Tutor(教师设计提示而非直接答案)在练习成绩 +127% 的同时,移除访问后的独立考试负效应基本消除(-0.004,不显著),说明精心设计的护栏可兼得练习提升与学习保持</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3544548.3580919</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">https://dl.acm.org/doi/10.1145/3544548.3580919</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3544548.3580919" title="Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming"><code>S-2023-kazemitabaar</code></a></td></tr><tr data-effect="negative" data-direction="negative" data-outcome="independent_problem_solving" data-search="e-004 study-bastani-2025 smpl-bastani-2025-n950 generative ai without guardrails can harm learning: evidence from high school mathematics 训练阶段使用 openai codex 的 10-17 岁新手在 45 道 python 代码编写任务上表现显著提升:完成率提高 1.15 倍、得分提高 1.8 倍 69 名 10-17 岁编程新手(含高中/初中年龄段) 训练阶段一半学习者可使用 openai codex 完成代码编写任务,任务后接代码修改任务 无 codex 访问组 negative s-2025-bastani"><td><code>E-004</code></td><td><strong>独立问题解决</strong></td><td><span class="dir neg">负向效应</span><span class="relation-note">与主张关系: 反驳</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>训练阶段使用 OpenAI Codex 的 10-17 岁新手在 45 道 Python 代码编写任务上表现显著提升:完成率提高 1.15 倍、得分提高 1.8 倍</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-BASTANI-2025</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-BASTANI-2025-N950</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>随机对照试验</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>high_school</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>69 名 10-17 岁编程新手(含高中/初中年龄段)</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>950</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>训练阶段一半学习者可使用 OpenAI Codex 完成代码编写任务,任务后接代码修改任务</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>无 Codex 访问组</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>exam without access to AI resources after practice phase</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>negative_17_percent_on_independent_exam</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>负向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>反驳</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>in_class_study_sessions</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>large-scale randomized controlled trial, practice phase then closed-book exam</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>large_scale_rct;independent_exam_without_ai;arm_wise_design</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>high_school_mathematics_not_university_programming;single_country</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>tool_design_difference</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1_study_design=2 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=1 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=partial_same_age_band_different_subject · subject_match=no_mathematics_vs_programming · tool_match=gpt4_chat_interface · scope=unguarded_general_chat_interface</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.75</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>训练阶段使用 OpenAI Codex 的 10-17 岁新手在 45 道 Python 代码编写任务上表现显著提升:完成率提高 1.15 倍、得分提高 1.8 倍</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://www.pnas.org/doi/10.1073/pnas.2422633122</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">https://www.pnas.org/doi/10.1073/pnas.2422633122</a></dd></div></dl></details></td><td><a class="source-link" href="https://www.pnas.org/doi/10.1073/pnas.2422633122" title="Generative AI without guardrails can harm learning: Evidence from high school mathematics"><code>S-2025-bastani</code></a></td></tr><tr data-effect="null" data-direction="null" data-outcome="independent_problem_solving" data-search="e-005 study-bastani-2025 smpl-bastani-2025-n950 generative ai without guardrails can harm learning: evidence from high school mathematics 训练期使用 codex 的学习者一周后评估后测成绩略好于对照组,但差异未达统计显著(保持力无显著差异);scratch 前测高分者若有 codex 使用史,保持后测显著更好 69 名 10-17 岁编程新手 训练阶段使用 openai codex 完成代码编写任务 无 codex 访问组;一周后评估后测 null s-2025-bastani"><td><code>E-005</code></td><td><strong>独立问题解决</strong></td><td><span class="dir neu">零效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>训练期使用 Codex 的学习者一周后评估后测成绩略好于对照组,但差异未达统计显著(保持力无显著差异);Scratch 前测高分者若有 Codex 使用史,保持后测显著更好</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-BASTANI-2025</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-BASTANI-2025-N950</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>随机对照试验</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>high_school</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>69 名 10-17 岁编程新手</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>950</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>训练阶段使用 OpenAI Codex 完成代码编写任务</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>无 Codex 访问组;一周后评估后测</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>exam without access to AI resources after practice phase</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>negative effect essentially eradicated, no positive effect observed</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>零效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>in_class_study_sessions</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>large-scale randomized controlled trial, three arms</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>direct_manipulation_of_tool_design;large_sample</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>no_positive_learning_gain_even_with_guardrails;subject_mismatch</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>prompt_engineering_effort</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1_study_design=2 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=1 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=partial · subject_match=no · tool_match=guardrailed_tutor_design · scope=guardrail_design_principle_transferable</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.75</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>训练期使用 Codex 的学习者一周后评估后测成绩略好于对照组,但差异未达统计显著(保持力无显著差异);Scratch 前测高分者若有 Codex 使用史,保持后测显著更好</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://www.pnas.org/doi/10.1073/pnas.2422633122</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">https://www.pnas.org/doi/10.1073/pnas.2422633122</a></dd></div></dl></details></td><td><a class="source-link" href="https://www.pnas.org/doi/10.1073/pnas.2422633122" title="Generative AI without guardrails can harm learning: Evidence from high school mathematics"><code>S-2025-bastani</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="assignment_score" data-search="e-006 study-bastani-2025 smpl-bastani-2025-n950 generative ai without guardrails can harm learning: evidence from high school mathematics 质性案例研究中,3 名 efl 学生珍视 chatgpt 的辅助价值(消除不确定性、澄清词汇、提供内容建议、语法/结构反馈,让学生专注于创意层面),并形成语言精修、观点生成与结构、校对与信心增强等使用策略 3 名不同水平(r1-r3)的 efl 学生,半结构化访谈 学生在学术写作过程中使用 chatgpt 的体验与策略(质性研究,无效应量测量) 无对照组(质性案例研究) positive s-2025-bastani"><td><code>E-006</code></td><td><strong>作业成绩</strong></td><td><span class="dir pos">正向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>质性案例研究中,3 名 EFL 学生珍视 ChatGPT 的辅助价值(消除不确定性、澄清词汇、提供内容建议、语法/结构反馈,让学生专注于创意层面),并形成语言精修、观点生成与结构、校对与信心增强等使用策略</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-BASTANI-2025</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-BASTANI-2025-N950</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>随机对照试验</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>high_school</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>3 名不同水平(R1-R3)的 EFL 学生,半结构化访谈</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>950</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>学生在学术写作过程中使用 ChatGPT 的体验与策略(质性研究,无效应量测量)</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>无对照组(质性案例研究)</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>practice problem performance during study sessions</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>48-127 percent improvement on practice problems</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>正向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>in_class_study_sessions</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>randomized controlled trial with practice and closed-book exam phases</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>same_study_compares_task_and_learning;large_sample</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>subject_mismatch_mathematics</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>task_familiarity</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1_study_design=2 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=1 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=partial · subject_match=no · tool_match=gpt4 · scope=task_performance_vs_learning_separation</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.75</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>质性案例研究中,3 名 EFL 学生珍视 ChatGPT 的辅助价值(消除不确定性、澄清词汇、提供内容建议、语法/结构反馈,让学生专注于创意层面),并形成语言精修、观点生成与结构、校对与信心增强等使用策略</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://www.pnas.org/doi/10.1073/pnas.2422633122</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">https://www.pnas.org/doi/10.1073/pnas.2422633122</a></dd></div></dl></details></td><td><a class="source-link" href="https://www.pnas.org/doi/10.1073/pnas.2422633122" title="Generative AI without guardrails can harm learning: Evidence from high school mathematics"><code>S-2025-bastani</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="knowledge_gain" data-search="e-007 study-marzuki-2024 smpl-marzuki-2024-n72 impact of chatgpt on esl students&#x27; academic writing skills 同一质性研究中,学生担忧 ai 使用的学术真实性与过度依赖风险(建议过于复杂/正式、语气不符、文化刻板印象等局限),强调必须保持人的判断并寻求教师/同伴反馈,呼吁伦理指引与批判性思维培养 3 名不同水平的 efl 学生,半结构化访谈 学生在学术写作过程中使用 chatgpt 的体验与策略 无对照组(质性案例研究) positive s-2024-marzuki"><td><code>E-007</code></td><td><strong>知识获得</strong></td><td><span class="dir pos">正向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">6</span><span class="quality-meter" aria-hidden="true"><i style="width:60%"></i></span></div></td><td class="claim-cell"><p>同一质性研究中,学生担忧 AI 使用的学术真实性与过度依赖风险(建议过于复杂/正式、语气不符、文化刻板印象等局限),强调必须保持人的判断并寻求教师/同伴反馈,呼吁伦理指引与批判性思维培养</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-MARZUKI-2024</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-MARZUKI-2024-N72</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Impact of ChatGPT on ESL students&#x27; academic writing skills</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Impact of ChatGPT on ESL students&#x27; academic writing skills</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2024</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>混合方法</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>undergraduate</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>3 名不同水平的 EFL 学生,半结构化访谈</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>72</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>学生在学术写作过程中使用 ChatGPT 的体验与策略</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>无对照组(质性案例研究)</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>writing tests with pre-post-delayed design</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>significant positive impact on writing skills</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>正向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>6_hours_intervention</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>mixed methods intervention study, pre/post/delayed tests and focus groups</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>delayed_post_test;mixed_methods_triangulation</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>short_intervention_6_hours;single_institution;elite_private_university</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>self_selection_consent</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1_study_design=1 · D2_sample_quality=1 · D3_measurement_validity=2 · D4_temporal_strength=2 · D5_directness=0</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>6.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=yes_undergraduate · subject_match=no_writing_not_programming · tool_match=chatgpt · scope=formative_feedback_writing</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>同一质性研究中,学生担忧 AI 使用的学术真实性与过度依赖风险(建议过于复杂/正式、语气不符、文化刻板印象等局限),强调必须保持人的判断并寻求教师/同伴反馈,呼吁伦理指引与批判性思维培养</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://link.springer.com/article/10.1186/s40561-024-00295-9</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://link.springer.com/article/10.1186/s40561-024-00295-9">https://link.springer.com/article/10.1186/s40561-024-00295-9</a></dd></div></dl></details></td><td><a class="source-link" href="https://link.springer.com/article/10.1186/s40561-024-00295-9" title="Impact of ChatGPT on ESL students&#x27; academic writing skills"><code>S-2024-marzuki</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="completion_time" data-search="e-008 study-peng-2023 smpl-peng-2023-n95 the impact of ai on developer productivity: evidence from github copilot 随机对照实验(n=95)显示:使用 copilot 的职业开发者完成标准化编程任务的用时比对照组缩短约 55%。 95 名经自由职业平台招募的职业开发者,完成标准化编码任务 任务期间可使用 github copilot 不可使用 copilot 的对照组 positive s-2023-peng"><td><code>E-008</code></td><td><strong>完成时间</strong></td><td><span class="dir pos">正向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>随机对照实验(n=95)显示:使用 Copilot 的职业开发者完成标准化编程任务的用时比对照组缩短约 55%。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-PENG-2023</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-PENG-2023-N95</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>The Impact of AI on Developer Productivity: Evidence from GitHub Copilot</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>The Impact of AI on Developer Productivity: Evidence from GitHub Copilot</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>随机对照试验</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>professional_developers_not_students</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>95 名经自由职业平台招募的职业开发者,完成标准化编码任务</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>95</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>任务期间可使用 GitHub Copilot</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>不可使用 Copilot 的对照组</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>time_to_complete_http_server_implementation</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>~55.8% faster task completion in Copilot group</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>正向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>single_task_session</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>online randomized controlled experiment with objective completion-time metric</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>randomized_controlled_design;objective_completion_time_metric</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>professional_population_not_students;single_task_ecology;preprint_not_peer_reviewed</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>task_familiarity;platform_recruitment_self_selection</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1_study_design=2 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=1 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=mismatch_professional_developers · subject_match=adjacent_web_development_task · tool_match=copilot_like_generative_ai · scope=task_performance_only_no_learning_outcome</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.6</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>随机对照实验(n=95)显示:使用 Copilot 的职业开发者完成标准化编程任务的用时比对照组缩短约 55%。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://doi.org/10.48550/arXiv.2302.06590</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://doi.org/10.48550/arXiv.2302.06590">https://doi.org/10.48550/arXiv.2302.06590</a></dd></div></dl></details></td><td><a class="source-link" href="https://doi.org/10.48550/arXiv.2302.06590" title="The Impact of AI on Developer Productivity: Evidence from GitHub Copilot"><code>S-2023-peng</code></a></td></tr><tr data-effect="null" data-direction="null" data-outcome="code_quality" data-search="e-009 study-yetistiren-2023 smpl-yetistiren-2023-bench github copilot ai pair programmer: asset or liability? 系统性基准评估显示 copilot 生成代码相对人类代码的质量结论不一:部分正确性具竞争力,同时记录到安全相关缺陷。 取自公开基准数据集的 copilot 生成程序与人类编写程序 copilot 生成的程序 相同基准上的人类编写程序 null s-2023-yetistiren"><td><code>E-009</code></td><td><strong>代码质量</strong></td><td><span class="dir neu">零效应</span><span class="relation-note">与主张关系: 中性</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>系统性基准评估显示 Copilot 生成代码相对人类代码的质量结论不一:部分正确性具竞争力,同时记录到安全相关缺陷。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-YETISTIREN-2023</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-YETISTIREN-2023-BENCH</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>GitHub Copilot AI Pair Programmer: Asset or Liability?</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>GitHub Copilot AI Pair Programmer: Asset or Liability?</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>观察性研究</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>not_applicable_code_artifacts</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>取自公开基准数据集的 Copilot 生成程序与人类编写程序</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>Copilot 生成的程序</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>相同基准上的人类编写程序</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>correctness_security_maintainability_metrics_on_benchmarks</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>mixed quality profile; no single-direction summary</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>零效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>中性</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>not_applicable_artifact_study</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>systematic empirical evaluation of generated code against human baselines on public benchmarks</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>multi_dimensional_quality_metrics;reproducible_benchmark_protocol</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>artifact_benchmark_not_classroom;no_learning_outcome;tool_version_from_2023</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>benchmark_task_distribution</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1_study_design=1 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=1 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=mismatch_no_learners_in_study · subject_match=introductory_adjacent_code_tasks · tool_match=copilot_like_generative_ai · scope=output_quality_only</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>系统性基准评估显示 Copilot 生成代码相对人类代码的质量结论不一:部分正确性具竞争力,同时记录到安全相关缺陷。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://doi.org/10.1016/j.jss.2023.111734</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://doi.org/10.1016/j.jss.2023.111734">https://doi.org/10.1016/j.jss.2023.111734</a></dd></div></dl></details></td><td><a class="source-link" href="https://doi.org/10.1016/j.jss.2023.111734" title="GitHub Copilot AI Pair Programmer: Asset or Liability?"><code>S-2023-yetistiren</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="assignment_score" data-search="e-010 study-finnieansley-2022 smpl-finnieansley-2022-qsets using github copilot to solve introductory programming problems codex 在 cs1 考试风格题目上能给出通过水平的解答(依数据集约 50%–75%),表明新手手中存在可观的任务能力余量。 公开 cs1 考试题集,并与已发表的学生分数分布比较 codex 对 cs1 题目作答生成 已发表的学生同届分数分布 positive s-2022-finnie-ansley"><td><code>E-010</code></td><td><strong>作业成绩</strong></td><td><span class="dir pos">正向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>Codex 在 CS1 考试风格题目上能给出通过水平的解答(依数据集约 50%–75%),表明新手手中存在可观的任务能力余量。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-FINNIEANSLEY-2022</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-FINNIEANSLEY-2022-QSETS</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Using GitHub Copilot to Solve Introductory Programming Problems</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Using GitHub Copilot to Solve Introductory Programming Problems</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2022</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>观察性研究</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>university_year_1_question_sets</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>公开 CS1 考试题集,并与已发表的学生分数分布比较</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>Codex 对 CS1 题目作答生成</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>已发表的学生同届分数分布</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>pass_rate_on_cs1_exam_style_questions</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>passing solutions on ~50-75% of questions across datasets</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>正向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>not_applicable_capability_probe</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>capability benchmark against published student distributions; reproducible question sets</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>public_reproducible_question_sets;directly_relevant_task_domain</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>tool_solves_task_does_not_equate_student_learning;codex_2021_model_version_outdated</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>question_leakage_into_training_data_possible</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1_study_design=1 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=1 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=partial_measures_tool_not_students · subject_match=introductory_programming · tool_match=copilot_like_generative_ai · scope=tool_capability_headroom</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>Codex 在 CS1 考试风格题目上能给出通过水平的解答(依数据集约 50%–75%),表明新手手中存在可观的任务能力余量。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3545945.3569830</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3545945.3569830">https://dl.acm.org/doi/10.1145/3545945.3569830</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3545945.3569830" title="Using GitHub Copilot to Solve Introductory Programming Problems"><code>S-2022-finnie-ansley</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="metacognition" data-search="e-011 study-explcomp-2023 smpl-explcomp-2023-ratings comparing code explanations created by students and large language models 受控比较发现 llm 生成的代码讲解与学生自撰讲解相当(部分更优),适合作为解释性支架材料,而非替代学生的解释练习。 同一批短程序的学生版与 llm 版讲解的受控对比 llm 生成的代码讲解 学生撰写的同题讲解 positive s-2023-explanations-compare"><td><code>E-011</code></td><td><strong>元认知</strong></td><td><span class="dir pos">正向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>受控比较发现 LLM 生成的代码讲解与学生自撰讲解相当(部分更优),适合作为解释性支架材料,而非替代学生的解释练习。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-EXPLCOMP-2023</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-EXPLCOMP-2023-RATINGS</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Comparing Code Explanations Created by Students and Large Language Models</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Comparing Code Explanations Created by Students and Large Language Models</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>观察性研究</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>university_introductory</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>同一批短程序的学生版与 LLM 版讲解的受控对比</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>LLM 生成的代码讲解</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>学生撰写的同题讲解</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>rated_explanation_quality_and_comprehensibility</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>comparable-or-better rated quality vs student explanations</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>正向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>single_session_ratings</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>controlled comparison with blind rating of explanation pairs</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>controlled_pairwise_comparison;learning_process_relevant_construct</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>short_term_ratings_not_learning_gains;small_program_snippets_ecology</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>rating_criteria_subjectivity</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1_study_design=1 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=1 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=partial_scaffold_material_only · subject_match=introductory_programming · tool_match=llm_explanations · scope=scaffold_quality_not_effectiveness</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>受控比较发现 LLM 生成的代码讲解与学生自撰讲解相当(部分更优),适合作为解释性支架材料,而非替代学生的解释练习。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3587102.3588785</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3587102.3588785">https://dl.acm.org/doi/10.1145/3587102.3588785</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3587102.3588785" title="Comparing Code Explanations Created by Students and Large Language Models"><code>S-2023-explanations-compare</code></a></td></tr><tr data-effect="negative" data-direction="negative" data-outcome="over_reliance" data-search="e-012 study-vaithilingam-2022 smpl-vaithilingam-2022-n24 expectation vs. experience: evaluating the usability of code generation tools 尽管首任务完成更快,参与者难以理解并调试 ai 生成的解法、对最终程序所有权感低——记录了纯速度指标遗漏的认知与依赖风险。 24 名参与者参与的组内设计可用性研究 copilot 类工具辅助编程 不使用工具的组内基线 negative s-2022-vaithilingam"><td><code>E-012</code></td><td><strong>过度依赖</strong></td><td><span class="dir neg">负向效应</span><span class="relation-note">与主张关系: 反驳</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>尽管首任务完成更快,参与者难以理解并调试 AI 生成的解法、对最终程序所有权感低——记录了纯速度指标遗漏的认知与依赖风险。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-VAITHILINGAM-2022</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-VAITHILINGAM-2022-N24</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Expectation vs. Experience: Evaluating the Usability of Code Generation Tools</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Expectation vs. Experience: Evaluating the Usability of Code Generation Tools</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2022</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>质性研究</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>mixed_cs_students_and_professionals</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>24 名参与者参与的组内设计可用性研究</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>24</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>Copilot 类工具辅助编程</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>不使用工具的组内基线</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>understanding_ownership_and_debugging_reports</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>documented comprehension/ownership difficulties despite speed gain</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>负向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>反驳</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>single_session</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>within-subject usability study with tasks, observation and interviews</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>rich_qualitative_process_data;constructs_missed_by_speed_metrics</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>small_n_24;single_session;self_reported_understanding</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>participant_ai_familiarity</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1_study_design=1 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=1 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=partial_includes_cs_students · subject_match=programming_adjacent · tool_match=copilot_like_generative_ai · scope=risk_identification</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>被反驳</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>尽管首任务完成更快,参与者难以理解并调试 AI 生成的解法、对最终程序所有权感低——记录了纯速度指标遗漏的认知与依赖风险。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3491101.3519665</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3491101.3519665">https://dl.acm.org/doi/10.1145/3491101.3519665</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3491101.3519665" title="Expectation vs. Experience: Evaluating the Usability of Code Generation Tools"><code>S-2022-vaithilingam</code></a></td></tr></tbody></table></div></div></section><section id="full-03-quality" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>03 证据可信度、反证与方法审计</h2><p class="full-chapter-lead">检查证据为什么可信、哪里冲突,以及哪些结论必须降级。</p></header><div class="full-chapter-body"><section class="method-review-group"><header class="method-review-title"><h3>审查目标:overall</h3><span class="method-verdict">关注</span></header><div class="method-audit-grid"><article class="method-audit-item method-met"><div class="method-audit-head"><strong>对照组</strong><span class="method-status">通过</span></div><p>证据集中所有提出因果主张的量化研究均含对照条件:Bastani(E-001/E-002/E-003)为三臂 RCT(无护栏 GPT Base / 护栏 GPT Tutor / 无 AI 对照组,N=2848);…</p><details class="method-detail detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>证据集中所有提出因果主张的量化研究均含对照条件:Bastani(E-001/E-002/E-003)为三臂 RCT(无护栏 GPT Base / 护栏 GPT Tutor / 无 AI 对照组,N=2848);Kazemitabaar(E-004/E-005)为随机对照(有/无 Codex,n=69);Shihab(E-011/E-012/E-013)为组内对照(同一批学生有/无 Copilot,n=10)。无对照的仅为不承担因果主张的质性/观察研究(Marzuki E-006/E-007 n=3 案例、Prather E-008/E-009 21 次观察性会话)与综述(Denny E-010),不构成对因果维度的威胁。但人群最匹配的实证研究(Prather、Marzuki)恰恰无对照,若下游把其主题当作效应证据则越界。</p></div></details></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>随机分配</strong><span class="method-status">通过</span></div><p>仅最大型的两项研究实施随机分配:Bastani(课堂内随机分组,约 1000 名学生,E-001/E-002/E-003)与 Kazemitabaar(随机对照,n=69,E-004/E-005)。…</p><details class="method-detail detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>仅最大型的两项研究实施随机分配:Bastani(课堂内随机分组,约 1000 名学生,E-001/E-002/E-003)与 Kazemitabaar(随机对照,n=69,E-004/E-005)。Shihab(E-011/E-012/E-013)为组内设计、无随机分配,叠加任务顺序与学习效应;Marzuki(E-006/E-007)与 Prather(E-008/E-009)非实验设计。随机化最充分的恰是人群最不匹配的研究(高中/数学),人群最相关的研究全部无随机化。</p></div></details></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>前测</strong><span class="method-status">通过</span></div><p>Kazemitabaar 明确报告 Scratch 前测并用于异质性分析(E-005:前测高分者保持后测显著更好);Bastani 以回归对照均值(0.284)控制基线(E-001),但证据记录未详述独立前测流程与组间基线等价性检验;…</p><details class="method-detail detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>Kazemitabaar 明确报告 Scratch 前测并用于异质性分析(E-005:前测高分者保持后测显著更好);Bastani 以回归对照均值(0.284)控制基线(E-001),但证据记录未详述独立前测流程与组间基线等价性检验;Shihab(E-011/E-012)以&#x27;高度相似任务&#x27;替代前测,无真实基线测量;Marzuki/Prather/Denny 不适用。期初基线控制整体不系统,正效应可能部分来自基线差异。</p></div></details></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>后测</strong><span class="method-status">通过</span></div><p>Bastani 设置移除 AI 后的独立考试(E-002:GPT Base 组 -17% 显著;E-003:Tutor 组 -0.004 不显著),是全集中设计最规范的后测;Kazemitabaar 有一周后延迟后测(E-005)。…</p><details class="method-detail detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>Bastani 设置移除 AI 后的独立考试(E-002:GPT Base 组 -17% 显著;E-003:Tutor 组 -0.004 不显著),是全集中设计最规范的后测;Kazemitabaar 有一周后延迟后测(E-005)。但 Shihab(E-011/E-012)仅在 AI 在场条件下测量任务表现,无无 AI 独立后测;Marzuki/Prather 无量化后测。正效应证据(E-001/E-004/E-011/E-012)全部缺乏&#x27;无 AI 独立后测&#x27;这一关键对照。</p></div></details></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>保持测试</strong><span class="method-status">部分</span></div><p>仅 Kazemitabaar 提供真正的延迟保持力测量(一周后评估后测,结果不显著,E-005);Bastani 的独立考试紧随移除 AI 之后(即时测量,E-002/E-003),测的是近迁移而非学期级保持;Shihab、Marzuki、Prather 均无保持力数据。…</p><details class="method-detail detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>仅 Kazemitabaar 提供真正的延迟保持力测量(一周后评估后测,结果不显著,E-005);Bastani 的独立考试紧随移除 AI 之后(即时测量,E-002/E-003),测的是近迁移而非学期级保持;Shihab、Marzuki、Prather 均无保持力数据。无任何研究覆盖与试点 16 周学期相当的时间尺度,长期保持(乃至期末、后续课程)证据完全空白。</p></div></details></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>迁移测试</strong><span class="method-status">部分</span></div><p>最有价值的迁移形式——从&#x27;AI 在场练习&#x27;迁移到&#x27;无 AI 独立表现&#x27;——在 Bastani(E-002/E-003)与 Kazemitabaar(E-005)中得到测量,这是试点成功判据(期末无 AI 机试/笔试)的直接对应证据。…</p><details class="method-detail detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>最有价值的迁移形式——从&#x27;AI 在场练习&#x27;迁移到&#x27;无 AI 独立表现&#x27;——在 Bastani(E-002/E-003)与 Kazemitabaar(E-005)中得到测量,这是试点成功判据(期末无 AI 机试/笔试)的直接对应证据。但两研究迁移距离有限(同领域同题型),无跨任务类型、跨课程(如数据结构)迁移测量;Shihab(E-011/E-012)在同一环境下切换工具条件,未测迁移;frame 的 secondary 追踪(后续课程表现)尚无证据基支撑其预期方向。</p></div></details></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>样本偏差</strong><span class="method-status">通过</span></div><p>样本构成与目标人群(国内大一零基础 C 语言新生)系统性错位:Bastani(E-001~E-003)为土耳其 9-11 年级高中生数学;Kazemitabaar(E-004/E-005)为 10-17 岁 K12 编程新手;Marzuki(E-006/E-007)为 EFL 学术写作;…</p><details class="method-detail detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>样本构成与目标人群(国内大一零基础 C 语言新生)系统性错位:Bastani(E-001~E-003)为土耳其 9-11 年级高中生数学;Kazemitabaar(E-004/E-005)为 10-17 岁 K12 编程新手;Marzuki(E-006/E-007)为 EFL 学术写作;唯一大学本科生样本 Shihab(E-011~E-013)仅 n=10 且任务为 brownfield Web 代码库(非零基础 C 入门);Prather(E-008/E-009)面向入门课程新手但为观察性研究。外推需跨&#x27;高中→大学&#x27;与&#x27;数学/Web→C 入门&#x27;两个维度,无单一研究同时匹配领域与年龄层。</p></div></details></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>自我选择偏差</strong><span class="method-status">部分</span></div><p>Bastani 为课堂内随机分组(E-001~E-003),自选偏倚最小,是全集最干净的样本;Kazemitabaar(E-004/E-005)系招募参加课后项目,存在一定自选;…</p><details class="method-detail detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>Bastani 为课堂内随机分组(E-001~E-003),自选偏倚最小,是全集最干净的样本;Kazemitabaar(E-004/E-005)系招募参加课后项目,存在一定自选;Shihab(E-011~E-013)n=10 全部自愿报名,志愿者通常更主动、技术接受度更高,正向速度/进展效应可能被高估;Marzuki(E-006/E-007)为目的性抽样/志愿者访谈样本,&#x27;珍视 AI 辅助价值&#x27;主题不能外推。自选偏倚与研究人群相关性成反比:越相关的样本自选越强。</p></div></details></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>测量效度</strong><span class="method-status">部分</span></div><p>完成率/正确率/速度作为&#x27;学习&#x27;代理的效度可疑:E-001 练习正确率、E-004 任务得分、E-011/E-012 完成速度与进展均在 AI 可访问条件下测得,AI 可直接产出答案抬高指标,无法区分&#x27;学会了&#x27;与&#x27;抄到了&#x27;(Bastani 机制数据 E-002:GPT Base 答对率 5…</p><details class="method-detail detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>完成率/正确率/速度作为&#x27;学习&#x27;代理的效度可疑:E-001 练习正确率、E-004 任务得分、E-011/E-012 完成速度与进展均在 AI 可访问条件下测得,AI 可直接产出答案抬高指标,无法区分&#x27;学会了&#x27;与&#x27;抄到了&#x27;(Bastani 机制数据 E-002:GPT Base 答对率 51% 中 42% 为逻辑错误;Wermelinger S-2023 显示 Copilot 可首次尝试解决 24 道典型入门题中的 16 道,FETCH_PARTIAL 仅验证到机构库摘要)。自评测量不可靠(E-002 学生过度乐观、E-008 能力错觉)。效度较高的测量(无 AI 独立考试 E-002/E-003、延迟后测 E-005)恰恰给出负向或零结果——测量选择本身决定结论方向。</p></div></details></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>混杂因素</strong><span class="method-status">部分</span></div><p>Bastani(E-001~E-003)三臂 RCT 控制最好(同一课堂环境、固定环节时长),但仍存在动机/参与度差异与护栏组教师设计提示带来的额外教学投入混淆。Shihab(E-011/E-012)组内设计叠加顺序效应、学习效应与疲劳(&#x27;高度相似任务&#x27;第二次完成必然更快);…</p><details class="method-detail detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>Bastani(E-001~E-003)三臂 RCT 控制最好(同一课堂环境、固定环节时长),但仍存在动机/参与度差异与护栏组教师设计提示带来的额外教学投入混淆。Shihab(E-011/E-012)组内设计叠加顺序效应、学习效应与疲劳(&#x27;高度相似任务&#x27;第二次完成必然更快);Kazemitabaar(E-004/E-005)课后项目环境存在练习量/监督差异。各研究均未系统控制练习时间、动机与同伴效应。</p></div></details></article><article class="method-audit-item method-not_applicable"><div class="method-audit-head"><strong>教师效应</strong><span class="method-status">不适用</span></div><p>证据记录未报告任何研究的教师/助教差异控制:Bastani 的护栏条件依赖&#x27;教师设计提示&#x27;(E-003),教师提示质量与护栏效果在设计中混淆,无法区分&#x27;护栏有效&#x27;与&#x27;好教师有效&#x27;;Kazemitabaar 课后项目可能单一导师;Shihab 同一研究者环境(教师恒定但样本单一)。…</p><details class="method-detail detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>证据记录未报告任何研究的教师/助教差异控制:Bastani 的护栏条件依赖&#x27;教师设计提示&#x27;(E-003),教师提示质量与护栏效果在设计中混淆,无法区分&#x27;护栏有效&#x27;与&#x27;好教师有效&#x27;;Kazemitabaar 课后项目可能单一导师;Shihab 同一研究者环境(教师恒定但样本单一)。对试点而言,护栏效果的可复制性取决于教师设计提示的能力,这一条件依赖未被任何研究分离检验。</p></div></details></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>新奇效应</strong><span class="method-status">部分</span></div><p>全部正效应来自短期干预,新奇效应不可排除:Bastani 仅 4 次 90 分钟课内环节(E-001 +48%/+127%,E-002 独立考试紧随其后)、Kazemitabaar 训练期任务(E-004)与一周后测(E-005)、Shihab 单次实验室会话(E-011/E-012)。…</p><details class="method-detail detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>全部正效应来自短期干预,新奇效应不可排除:Bastani 仅 4 次 90 分钟课内环节(E-001 +48%/+127%,E-002 独立考试紧随其后)、Kazemitabaar 训练期任务(E-004)与一周后测(E-005)、Shihab 单次实验室会话(E-011/E-012)。无任何研究覆盖 16 周学期周期,新奇感消退后的长期参与度与学习效应维持无证据;skeptic 的 novelty_effect 检查亦为 medium 风险。</p></div></details></article><article class="method-audit-item method-not_applicable"><div class="method-audit-head"><strong>工具版本效应</strong><span class="method-status">不适用</span></div><p>证据工具代际与试点工具不同代:Kazemitabaar(E-004/E-005)为 2023 年 OpenAI Codex(code-davinci 时代)、Marzuki(E-006/E-007)为 GPT-3.5/4 时代 ChatGPT、Bastani(E-001~E-003)为 G…</p><details class="method-detail detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>证据工具代际与试点工具不同代:Kazemitabaar(E-004/E-005)为 2023 年 OpenAI Codex(code-davinci 时代)、Marzuki(E-006/E-007)为 GPT-3.5/4 时代 ChatGPT、Bastani(E-001~E-003)为 GPT-4 时代、Shihab(E-011~E-013)为 2024-2025 Copilot;试点学年(2025-2026)学生面对 GPT-5 级模型与 IDE 深度集成智能体,能力更强、集成更深,收益与依赖风险可能同时放大,2023-2025 效应量不能直接外推。</p></div></details></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>AI 使用规则</strong><span class="method-status">部分</span></div><p>Bastani(E-001~E-003)是唯一直接操纵 AI 使用政策的研究:无护栏 GPT Base(类标准 ChatGPT 界面,可抄答案)vs 护栏 GPT Tutor(教师设计提示、不给直接答案),对应实证了&#x27;允许使用但无规则→独立考试 -17% 伤害&#x27;与&#x27;有护栏→练习 +127%…</p><details class="method-detail detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>Bastani(E-001~E-003)是唯一直接操纵 AI 使用政策的研究:无护栏 GPT Base(类标准 ChatGPT 界面,可抄答案)vs 护栏 GPT Tutor(教师设计提示、不给直接答案),对应实证了&#x27;允许使用但无规则→独立考试 -17% 伤害&#x27;与&#x27;有护栏→练习 +127% 且负效应消除&#x27;的政策对比,与试点&#x27;禁止直接提交 AI 代码、实验课独立评测&#x27;的护栏设计同构。局限:护栏效果仅在单一情境(高中数学、教师设计提示)验证过,需在 C 语言场景复验;且各研究均未验证对照组依从性(对照组成员是否实际未使用 AI),政策污染未排除。</p></div></details></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>样本流失</strong><span class="method-status">部分</span></div><p>证据记录未报告各研究的退出率、缺失数据处理与依从性验证:Bastani 约 1000 名学生、4 次环节、2848 观测(E-001~E-003),跨环节流失与缺失未说明;Kazemitabaar n=69 一周后测(E-005)的保留率未报告;…</p><details class="method-detail detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>证据记录未报告各研究的退出率、缺失数据处理与依从性验证:Bastani 约 1000 名学生、4 次环节、2848 观测(E-001~E-003),跨环节流失与缺失未说明;Kazemitabaar n=69 一周后测(E-005)的保留率未报告;Shihab n=10 完成全部任务并参加退出访谈(E-011~E-013),是唯一可推断低退出的研究;Marzuki n=3 访谈样本。意图治疗 vs 实际使用(对照组偷偷用 AI)的依从性分析在全集中缺失。</p></div></details></article></div><div class="method-guard-wrap"><strong>任务 vs 学习护栏:</strong><div class="method-guard"><p>证据集本身未把任务表现等同学习效果:三类独立测量(E-002/E-003/E-005)与任务表现测量(E-001/E-004/E-011/E-012)被明确分开,且独立测量给出负向/零结果,恰是对照铁律(SKILL.md RULE 3:task performance 不得自动等同 learning effect)的正确执行。…</p><details class="detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>证据集本身未把任务表现等同学习效果:三类独立测量(E-002/E-003/E-005)与任务表现测量(E-001/E-004/E-011/E-012)被明确分开,且独立测量给出负向/零结果,恰是对照铁律(SKILL.md RULE 3:task performance 不得自动等同 learning effect)的正确执行。但风险在边界处:(a) 若下游综合以 E-001 的 +127% 或 E-011 的快 35% 作为&#x27;学习提升&#x27;证据,即违反铁律,证据天平会系统性偏向&#x27;允许使用&#x27;;(b) E-001 练习成绩提升与 E-002 独立考试伤害在同一研究中并存,任何只引其一的做法都会误导。frame 的 outcomes 已把&#x27;任务表现&#x27;与&#x27;学习能力&#x27;分开测量、并把期末无 AI 统一机试/笔试设为唯一成功判据,与本 guard 一致。</p></div></details></div></div></section><article class="conflict-card"><strong>裁决说明:</strong><p class="conflict-text">分歧来自结果分离(任务 vs 学习)、工具设计(有护栏 vs 无护栏)与人群(K-12/职业者 vs 大学生)。随机实验与基准研究中任务表现证据一致为正;唯一测量移除 AI 后独立表现的研究显示无护栏时有害;可用性与工件研究补充依赖与质量警示而非解决学习问题。</p></article><div class="evidence-tribunal" data-visual="evidence-tribunal-grid"><p class="tribunal-summary"><strong>决策:</strong>试点验证 · <strong>置信度:</strong>中</p><div class="tribunal-grid"><article class="tribunal-card supported"><header><span aria-hidden="true">✓</span><h3>可以主张</h3><span class="tribunal-count">7</span></header><ul><li><p>AI 编程助手在训练期间可靠地提升任务表现(完成速度、正确性)—— E-001、E-006。</p><div class="tribunal-evidence-refs"><code>E-001</code><code>E-006</code></div></li><li><p>无护栏的生成式 AI 访问在移除工具后可能损害独立问题解决能力 —— E-004。</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li><li><p>护栏设计(给提示而非给答案)能大幅缓解负面学习效应 —— E-005。</p><div class="tribunal-evidence-refs"><code>E-005</code></div></li><li><p>任务表现提升并不自动等于学习提升 —— E-004 与 E-006 的研究内对照。</p><div class="tribunal-evidence-refs"><code>E-004</code><code>E-006</code></div></li><li><p>工具能力可观:Codex 能解出约半数至四分之三的 CS1 考试风格题目 —— E-010。</p><div class="tribunal-evidence-refs"><code>E-010</code></div></li><li><p>职业开发者 RCT 显示 Copilot 带来约 55% 任务提速;但职业人群限制直接性 —— E-008。</p><div class="tribunal-evidence-refs"><code>E-008</code></div></li><li><p>LLM 代码讲解的质量评级与学生自撰讲解相当,可作支架材料 —— E-011。</p><div class="tribunal-evidence-refs"><code>E-011</code></div></li></ul></article><article class="tribunal-card uncertain"><header><span aria-hidden="true">?</span><h3>尚不能主张</h3><span class="tribunal-count">4</span></header><ul><li><p>AI 编程助手能否真正改善或保持大学新手的编程学习——本证据集中没有大学层面的直接 RCT [无直接证据]</p></li><li><p>Kazemitabaar 2023 的一周中性保持性能否延伸到一个学期 —— E-003。</p><div class="tribunal-evidence-refs"><code>E-003</code></div></li><li><p>基准质量结论(E-009)与讲解质量评级(E-011)能否转化为课堂学习收益。</p><div class="tribunal-evidence-refs"><code>E-009</code><code>E-011</code></div></li><li><p>可用性研究所记录的理解/所有权困难(E-012)在整学期护栏条件下会如何演变。</p><div class="tribunal-evidence-refs"><code>E-012</code></div></li></ul></article><article class="tribunal-card contradicted"><header><span aria-hidden="true">×</span><h3>被反驳的主张</h3><span class="tribunal-count">2</span></header><ul><li><p>&#x27;AI 工具总能提高学习&#x27;被 E-004 反驳(无护栏访问,独立考试 −17%)。</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li><li><p>&#x27;速度收益等于学习收益&#x27;被 E-001/E-006/E-008 与 E-004 之间的任务-学习分离所反驳。</p><div class="tribunal-evidence-refs"><code>E-001</code><code>E-006</code><code>E-008</code><code>E-004</code></div></li></ul></article><article class="tribunal-card missing"><header><span aria-hidden="true">…</span><h3>缺失证据</h3><span class="tribunal-count">4</span></header><ul><li><p>在大学编程课程中带保持与无 AI 迁移测试的 RCT。</p></li><li><p>同一课程内变化 AI 使用政策的研究。</p></li><li><p>跨越一门课的 AI 依赖纵向数据。</p></li><li><p>职业提速 RCT 的同行评审重复(Peng 等仍为预印本)。</p></li></ul></article></div><details class="supporting-visual"><summary>EvidenceFlow 协议</summary><svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="EvidenceFlow 协议流程"><title>EvidenceFlow 协议流程</title><desc>从问题框架、检索、抓取验证、证据抽取、反方质疑、方法审计、裁决到适用性与干预评价的完整流程。</desc><defs><marker id="arr-zh-full-workflow" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">EvidenceFlow 协议</text><rect x="24" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="56.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="56.0" y="126.0">问题框架</tspan></text><line x1="88" y1="126" x2="100" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="100" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="132.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="132.0" y="126.0">检索</tspan></text><line x1="164" y1="126" x2="176" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="176" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="208.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="208.0" y="126.0">抓取</tspan><tspan x="208.0" y="140.0">验证</tspan></text><line x1="240" y1="126" x2="252" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="252" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="284.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="284.0" y="126.0">证据抽取</tspan></text><line x1="316" y1="126" x2="328" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="328" y="100" width="64" height="52" rx="8" fill="#5E8A6A"/><text x="360.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="360.0" y="126.0">反方质疑</tspan></text><line x1="392" y1="126" x2="404" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="404" y="100" width="64" height="52" rx="8" fill="#5E8A6A"/><text x="436.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="436.0" y="126.0">方法</tspan><tspan x="436.0" y="140.0">审计</tspan></text><line x1="468" y1="126" x2="480" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="480" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="512.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="512.0" y="126.0">裁决</tspan></text><line x1="544" y1="126" x2="556" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="556" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="588.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="588.0" y="126.0">适用性</tspan></text><line x1="620" y1="126" x2="632" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="632" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="664.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="664.0" y="126.0">干预</tspan><tspan x="664.0" y="140.0">评价</tspan></text></svg></details><details class="supporting-visual"><summary>裁决信息图</summary><svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="证据裁决信息图"><title>证据裁决信息图</title><desc>可以主张与不可主张的证据 ID 与建议决策徽章;完整主张文本见下方裁决卡片。</desc><defs><marker id="arr-zh-full-tribunal" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">证据裁决</text><text x="24" y="180" font-size="13" font-weight="700" fill="#3A3833">冲突来源</text><text x="24" y="202" font-size="11" fill="#8A867E">详见下方裁决卡片</text><rect x="24" y="216" width="180" height="28" rx="14" fill="#C99A4A"/><text x="114" y="235" text-anchor="middle" font-size="13" font-weight="700" fill="#fff">PILOT</text><text x="250" y="80" font-size="13" font-weight="700" fill="#5E8A6A">可以主张 (7)</text><circle cx="254" cy="100" r="3" fill="#5E8A6A"/><text x="266" y="105" font-size="11" fill="#3A3833">E-001</text><circle cx="254" cy="122" r="3" fill="#5E8A6A"/><text x="266" y="127" font-size="11" fill="#3A3833">E-006</text><circle cx="254" cy="144" r="3" fill="#5E8A6A"/><text x="266" y="149" font-size="11" fill="#3A3833">E-004</text><circle cx="254" cy="166" r="3" fill="#5E8A6A"/><text x="266" y="171" font-size="11" fill="#3A3833">E-005</text><text x="490" y="80" font-size="13" font-weight="700" fill="#A85B53">不可主张 (2)</text><circle cx="494" cy="100" r="3" fill="#A85B53"/><text x="506" y="105" font-size="11" fill="#3A3833">E-004</text><circle cx="494" cy="122" r="3" fill="#A85B53"/><text x="506" y="127" font-size="11" fill="#3A3833">E-001</text><circle cx="494" cy="144" r="3" fill="#A85B53"/><text x="506" y="149" font-size="11" fill="#3A3833">E-006</text><circle cx="494" cy="166" r="3" fill="#A85B53"/><text x="506" y="171" font-size="11" fill="#3A3833">E-008</text></svg></details></div><div class="trace-chain" data-visual="trace-chain"><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-001</strong><p class="trace-claim-text">在练习环节,无护栏的 GPT Base(类标准 ChatGPT 界面)使高中生的练习成绩相对对照组提高 48%,带护栏的 GPT Tutor 提高 127%(Table 1:practice 系数 0.137/0.361,对照均值 0.284)</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-001</code><span>完成时间</span><span class="dir pos">正向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 9.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3544548.3580919"><code>S-2023-kazemitabaar</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-002</strong><p class="trace-claim-text">移除 AI 访问后的无辅助独立考试中,GPT Base 组成绩比从未使用 AI 的对照组低 17%(统计显著),表明无护栏使用 AI 损害技能习得;机制上学生把 GPT 当&#x27;拐杖&#x27;直接抄答案(GPT Base 答对率仅 51%,其中 42% 逻辑错误、8% 算术错误),且学生自评过度乐观、未察觉学习受损</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-002</code><span>独立问题解决</span><span class="dir neu">零效应</span><span class="trace-relation">主张关系: 中性</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3544548.3580919"><code>S-2023-kazemitabaar</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-003</strong><p class="trace-claim-text">带护栏的 GPT Tutor(教师设计提示而非直接答案)在练习成绩 +127% 的同时,移除访问后的独立考试负效应基本消除(-0.004,不显著),说明精心设计的护栏可兼得练习提升与学习保持</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-003</code><span>记忆保持</span><span class="dir neu">零效应</span><span class="trace-relation">主张关系: 中性</span><span>Q 9.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3544548.3580919"><code>S-2023-kazemitabaar</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-004</strong><p class="trace-claim-text">训练阶段使用 OpenAI Codex 的 10-17 岁新手在 45 道 Python 代码编写任务上表现显著提升:完成率提高 1.15 倍、得分提高 1.8 倍</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-004</code><span>独立问题解决</span><span class="dir neg">负向效应</span><span class="trace-relation">主张关系: 反驳</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122"><code>S-2025-bastani</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-005</strong><p class="trace-claim-text">训练期使用 Codex 的学习者一周后评估后测成绩略好于对照组,但差异未达统计显著(保持力无显著差异);Scratch 前测高分者若有 Codex 使用史,保持后测显著更好</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-005</code><span>独立问题解决</span><span class="dir neu">零效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122"><code>S-2025-bastani</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-006</strong><p class="trace-claim-text">质性案例研究中,3 名 EFL 学生珍视 ChatGPT 的辅助价值(消除不确定性、澄清词汇、提供内容建议、语法/结构反馈,让学生专注于创意层面),并形成语言精修、观点生成与结构、校对与信心增强等使用策略</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-006</code><span>作业成绩</span><span class="dir pos">正向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122"><code>S-2025-bastani</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-007</strong><p class="trace-claim-text">同一质性研究中,学生担忧 AI 使用的学术真实性与过度依赖风险(建议过于复杂/正式、语气不符、文化刻板印象等局限),强调必须保持人的判断并寻求教师/同伴反馈,呼吁伦理指引与批判性思维培养</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-007</code><span>知识获得</span><span class="dir pos">正向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 6.0</span><span class="trace-arrow">→</span><a href="https://link.springer.com/article/10.1186/s40561-024-00295-9"><code>S-2024-marzuki</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-008</strong><p class="trace-claim-text">随机对照实验(n=95)显示:使用 Copilot 的职业开发者完成标准化编程任务的用时比对照组缩短约 55%。</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-008</code><span>完成时间</span><span class="dir pos">正向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://doi.org/10.48550/arXiv.2302.06590"><code>S-2023-peng</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-009</strong><p class="trace-claim-text">系统性基准评估显示 Copilot 生成代码相对人类代码的质量结论不一:部分正确性具竞争力,同时记录到安全相关缺陷。</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-009</code><span>代码质量</span><span class="dir neu">零效应</span><span class="trace-relation">主张关系: 中性</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://doi.org/10.1016/j.jss.2023.111734"><code>S-2023-yetistiren</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-010</strong><p class="trace-claim-text">Codex 在 CS1 考试风格题目上能给出通过水平的解答(依数据集约 50%–75%),表明新手手中存在可观的任务能力余量。</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-010</code><span>作业成绩</span><span class="dir pos">正向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3545945.3569830"><code>S-2022-finnie-ansley</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-011</strong><p class="trace-claim-text">受控比较发现 LLM 生成的代码讲解与学生自撰讲解相当(部分更优),适合作为解释性支架材料,而非替代学生的解释练习。</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-011</code><span>元认知</span><span class="dir pos">正向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3587102.3588785"><code>S-2023-explanations-compare</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-012</strong><p class="trace-claim-text">尽管首任务完成更快,参与者难以理解并调试 AI 生成的解法、对最终程序所有权感低——记录了纯速度指标遗漏的认知与依赖风险。</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-012</code><span>过度依赖</span><span class="dir neg">负向效应</span><span class="trace-relation">主张关系: 反驳</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3491101.3519665"><code>S-2022-vaithilingam</code></a></div></div></article></div><div id="chart-trace-zh" class="chart-mount" aria-label="主张-证据追溯"></div><p class="chart-interpretation"><strong>这意味着什么:</strong>每个重要主张都必须能追到 Evidence ID 和原始来源。</p></div></section><section id="full-04-action" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>04 适用范围与教学行动</h2><p class="full-chapter-lead">把可外推范围、护栏和教学动作连接到具体证据。</p></header><div class="full-chapter-body"><p><strong>目标人群:</strong>首次学习 C 语言编程的大一计算机专业学生</p>
2187
+ </div><div class="scope-grid"><article class="scope-card"><h3>研究问题</h3><p class="scope-text">我们准备在大学一年级 C 语言课程中允许学生使用生成式 AI 编程助手。它到底会不会提高学习效果?应该怎样引入?</p></article><article class="scope-card"><h3>目标学习者</h3><p class="scope-text">教育阶段:大学一年级;专业:计算机科学与技术;先验知识:首次程序设计课程,无文本编程基础;学习者特征:混合能力大班(60 人)</p></article><article class="scope-card"><h3>课程情境</h3><p class="scope-text">课程:C 语言程序设计;课程类型:讲授 + 实验课;课程周期:16 周(一学期)</p></article><article class="scope-card"><h3>AI 干预</h3><p class="scope-text">干预方式:讲授 + 实验练习;AI 工具:生成式 AI 编程助手;允许使用:设计中(待证据评审);使用频率:每周实验课;干预周期:一学期</p></article><article class="scope-card"><h3>比较条件</h3><p class="scope-text">No AI Coding Assistant Control</p></article><article class="scope-card"><h3>结果构念</h3><p class="scope-text">主要结果:独立问题解决、代码质量;次要结果:完成时间、记忆保持、知识获得;风险结果:AI 依赖、过度依赖、迁移下降</p></article><article class="scope-card"><h3>研究范围</h3><p class="scope-text">时间范围:2021 2026;地域:Worldwide;研究设计:随机对照试验、准实验、观察性研究</p></article><article class="scope-card"><h3>决策成功条件</h3><p class="scope-text">independent problem solving and code quality improve (or do not decline) while AI dependency risk stays controlled; evidence base supports a bounded pilot.</p></article></div></div></section><section id="full-02-evidence" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>02 关键证据与结果分离</h2><p class="full-chapter-lead">把任务表现、真实学习、保持与风险放在同一证据地图中,但不混为一谈。</p></header><div class="full-chapter-body"><div class="retrieval-grid"><article><h3>纳入标准</h3><ul><li>Studies Of Generative AI Coding Tools In Learning To Program</li><li>Outcomes Measuring Learning Not Only Task Speed</li><li>University Or Novice Programming Populations</li></ul></article><article><h3>排除标准</h3><ul><li>Practitioner Anecdotes Without Data</li><li>Industry Professional Populations Only</li></ul></article></div><div class="retrieval-coverage"><h3>证据来源覆盖</h3><p><code>S-2023-kazemitabaar</code> <code>S-2025-bastani</code> <code>S-2024-marzuki</code> <code>S-2023-peng</code> <code>S-2023-yetistiren</code> <code>S-2022-finnie-ansley</code> <code>S-2023-explanations-compare</code> <code>S-2022-vaithilingam</code></p><p class="retrieval-note">当前报告只展示 result 中真实存在的检索与来源信息;没有流程计数时不伪造 PRISMA / funnel 数字。</p></div><div class="outcome-separation" data-visual="outcome-separation"><div class="visual-heading"><h3>结果分离 · 任务表现 ≠ 学习效果</h3><p>将不同结果类型分开裁决,避免把训练时更快、更高分直接等同于真正学会。</p><p class="semantic-note">效应方向来自 evidence.effect_direction;“支持某个主张”不等于“结果是正向”。</p></div><div class="outcome-groups"><article class="outcome-group outcome-task"><h3>任务 / 近端表现</h3><ul><li><strong>完成时间</strong><span class="outcome-states"><span class="dir pos">正向效应 2</span></span></li><li><strong>代码质量</strong><span class="outcome-states"><span class="dir neu">零效应 1</span></span></li><li><strong>作业成绩</strong><span class="outcome-states"><span class="dir pos">正向效应 2</span></span></li></ul></article><article class="outcome-group outcome-learning"><h3>学习 / 保持 / 迁移</h3><ul><li><strong>知识获得</strong><span class="outcome-states"><span class="dir pos">正向效应 1</span></span></li><li><strong>记忆保持</strong><span class="outcome-states"><span class="dir neu">零效应 1</span></span></li><li><strong>独立问题解决</strong><span class="outcome-states"><span class="dir neg">负向效应 1</span><span class="dir neu">零效应 2</span></span></li></ul></article><article class="outcome-group outcome-risk"><h3>风险 / 依赖</h3><ul><li><strong>过度依赖</strong><span class="outcome-states"><span class="dir neg">负向效应 1</span></span></li></ul></article><article class="outcome-group outcome-other"><h3>其他结果</h3><ul><li><strong>元认知</strong><span class="outcome-states"><span class="dir pos">正向效应 1</span></span></li></ul></article></div></div><div class='table-wrap outcome-table'><table class='data-table'><thead><tr><th>结果类型</th><th>正向效应</th><th>负向效应</th><th>零效应</th><th>证据</th></tr></thead><tbody><tr><td><strong>知识获得</strong><span class='raw-tag' title='原始标识'>knowledge_gain</span></td><td class='num'>1</td><td class='num'>0</td><td class='num'>0</td><td><code>E-007</code> </td></tr><tr><td><strong>记忆保持</strong><span class='raw-tag' title='原始标识'>retention</span></td><td class='num'>0</td><td class='num'>0</td><td class='num'>1</td><td><code>E-003</code> </td></tr><tr><td><strong>独立问题解决</strong><span class='raw-tag' title='原始标识'>independent_problem_solving</span></td><td class='num'>0</td><td class='num'>1</td><td class='num'>2</td><td><code>E-002</code> <code>E-004</code> <code>E-005</code> </td></tr><tr><td><strong>完成时间</strong><span class='raw-tag' title='原始标识'>completion_time</span></td><td class='num'>2</td><td class='num'>0</td><td class='num'>0</td><td><code>E-001</code> <code>E-008</code> </td></tr><tr><td><strong>代码质量</strong><span class='raw-tag' title='原始标识'>code_quality</span></td><td class='num'>0</td><td class='num'>0</td><td class='num'>1</td><td><code>E-009</code> </td></tr><tr><td><strong>作业成绩</strong><span class='raw-tag' title='原始标识'>assignment_score</span></td><td class='num'>2</td><td class='num'>0</td><td class='num'>0</td><td><code>E-006</code> <code>E-010</code> </td></tr><tr><td><strong>元认知</strong><span class='raw-tag' title='原始标识'>metacognition</span></td><td class='num'>1</td><td class='num'>0</td><td class='num'>0</td><td><code>E-011</code> </td></tr><tr><td><strong>过度依赖</strong><span class='raw-tag' title='原始标识'>over_reliance</span></td><td class='num'>0</td><td class='num'>1</td><td class='num'>0</td><td><code>E-012</code> </td></tr></tbody></table></div><div class="visual-surface" data-visual="outcome-evidence-balance"><svg viewBox="0 0 720 300" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="各结果类型证据效应分布"><title>各结果类型证据效应分布</title><desc>各结果类型的正向 / 负向 / 零效应证据条数(基于 effect_direction,不等同于 Claim 是否被支持)。</desc><rect x="0" y="0" width="720" height="300" fill="#FFFFFF"/><line x1="415.0" y1="46" x2="415.0" y2="182" stroke="#999" stroke-width="1" stroke-dasharray="3,3"/><text x="142" y="58.0" text-anchor="end" font-size="11" fill="#333">知识获得</text><rect x="415.0" y="49.8" width="132.5" height="9.4" fill="#5E8A6A"/><text x="481.2" y="58.0" text-anchor="middle" font-size="9" fill="#fff">1</text><text x="142" y="75.0" text-anchor="end" font-size="11" fill="#333">记忆保持</text><text x="142" y="92.0" text-anchor="end" font-size="11" fill="#333">独立问题解决</text><rect x="282.5" y="83.8" width="132.5" height="9.4" fill="#A85B53"/><text x="348.8" y="92.0" text-anchor="middle" font-size="9" fill="#fff">-1</text><text x="142" y="109.0" text-anchor="end" font-size="11" fill="#333">完成时间</text><rect x="415.0" y="100.8" width="265.0" height="9.4" fill="#5E8A6A"/><text x="547.5" y="109.0" text-anchor="middle" font-size="9" fill="#fff">2</text><text x="142" y="126.0" text-anchor="end" font-size="11" fill="#333">代码质量</text><text x="142" y="143.0" text-anchor="end" font-size="11" fill="#333">作业成绩</text><rect x="415.0" y="134.8" width="265.0" height="9.4" fill="#5E8A6A"/><text x="547.5" y="143.0" text-anchor="middle" font-size="9" fill="#fff">2</text><text x="142" y="160.0" text-anchor="end" font-size="11" fill="#333">元认知</text><rect x="415.0" y="151.8" width="132.5" height="9.4" fill="#5E8A6A"/><text x="481.2" y="160.0" text-anchor="middle" font-size="9" fill="#fff">1</text><text x="142" y="177.0" text-anchor="end" font-size="11" fill="#333">过度依赖</text><rect x="282.5" y="168.8" width="132.5" height="9.4" fill="#A85B53"/><text x="348.8" y="177.0" text-anchor="middle" font-size="9" fill="#fff">-1</text><rect x="415.0" y="206.1" width="8.0" height="6" fill="#C99A4A"/><rect x="415.0" y="214.9" width="8.0" height="6" fill="#C99A4A"/><rect x="415.0" y="232.4" width="8.0" height="6" fill="#C99A4A"/><rect x="150" y="14" width="10" height="10" fill="#5E8A6A"/><text x="164" y="23" font-size="10" fill="#333">正向效应</text><rect x="226" y="14" width="10" height="10" fill="#A85B53"/><text x="240" y="23" font-size="10" fill="#333">负向效应</text><rect x="302" y="14" width="10" height="10" fill="#C99A4A"/><text x="316" y="23" font-size="10" fill="#333">零效应</text></svg><p class="chart-interpretation"><strong>这意味着什么:</strong>各 Outcome 的正向 / 负向 / 零效应证据条数对比。这里展示的是 effect_direction,不是“这条证据是否支持某个主张”。</p></div><div id="chart-outcome-zh" class="chart-mount" aria-label="结果证据概览"></div><figure class="academic-figure" data-visual="outcome-evidence-balance"><svg viewBox="0 0 923 300" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="各结果类型效应方向分布(出版级学术图)"><title>各结果类型效应方向分布(出版级学术图)</title><desc>各结果类型的正向 / 负向 / 零效应证据条数;计数轴整数刻度,不随主题变化。来源:EduEvidence result.json。</desc><rect width="923" height="300" fill="#FFFFFF"/><line x1="70" y1="250" x2="650" y2="250" stroke="#333" stroke-width="1"/><text x="106.2" y="266" text-anchor="middle" font-size="10" fill="#333">知识获得</text><rect x="88.1" y="150.0" width="18.1" height="100.0" fill="#38BDF8"/><text x="97.2" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><rect x="106.2" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="124.4" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><text x="178.8" y="266" text-anchor="middle" font-size="10" fill="#333">记忆保持</text><rect x="160.6" y="250.0" width="18.1" height="1.0" fill="#38BDF8"/><rect x="178.8" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="196.9" y="150.0" width="18.1" height="100.0" fill="#F59E0B"/><text x="205.9" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><text x="251.2" y="266" text-anchor="middle" font-size="10" fill="#333">独立问题解决</text><rect x="233.1" y="250.0" width="18.1" height="1.0" fill="#38BDF8"/><rect x="251.2" y="150.0" width="18.1" height="100.0" fill="#10B981"/><text x="260.3" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><rect x="269.4" y="50.0" width="18.1" height="200.0" fill="#F59E0B"/><text x="278.4" y="47.0" text-anchor="middle" font-size="9" fill="#333">2</text><text x="323.8" y="266" text-anchor="middle" font-size="10" fill="#333">完成时间</text><rect x="305.6" y="50.0" width="18.1" height="200.0" fill="#38BDF8"/><text x="314.7" y="47.0" text-anchor="middle" font-size="9" fill="#333">2</text><rect x="323.8" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="341.9" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><text x="396.2" y="266" text-anchor="middle" font-size="10" fill="#333">代码质量</text><rect x="378.1" y="250.0" width="18.1" height="1.0" fill="#38BDF8"/><rect x="396.2" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="414.4" y="150.0" width="18.1" height="100.0" fill="#F59E0B"/><text x="423.4" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><text x="468.8" y="266" text-anchor="middle" font-size="10" fill="#333">作业成绩</text><rect x="450.6" y="50.0" width="18.1" height="200.0" fill="#38BDF8"/><text x="459.7" y="47.0" text-anchor="middle" font-size="9" fill="#333">2</text><rect x="468.8" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="486.9" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><text x="541.2" y="266" text-anchor="middle" font-size="10" fill="#333">元认知</text><rect x="523.1" y="150.0" width="18.1" height="100.0" fill="#38BDF8"/><text x="532.2" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><rect x="541.2" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="559.4" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><text x="613.8" y="266" text-anchor="middle" font-size="10" fill="#333">过度依赖</text><rect x="595.6" y="250.0" width="18.1" height="1.0" fill="#38BDF8"/><rect x="613.8" y="150.0" width="18.1" height="100.0" fill="#10B981"/><text x="622.8" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><rect x="631.9" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><line x1="65" y1="250.0" x2="70" y2="250.0" stroke="#999"/><text x="62" y="253.0" text-anchor="end" font-size="9" fill="#666">0</text><line x1="65" y1="150.0" x2="70" y2="150.0" stroke="#999"/><text x="62" y="153.0" text-anchor="end" font-size="9" fill="#666">1</text><line x1="65" y1="50.0" x2="70" y2="50.0" stroke="#999"/><text x="62" y="53.0" text-anchor="end" font-size="9" fill="#666">2</text><text x="360.0" y="30" text-anchor="middle" font-size="14" font-weight="700" fill="#111">各结果类型的效应方向分布</text><rect x="70" y="8" width="10" height="10" fill="#38BDF8"/><text x="84" y="17" font-size="10" fill="#333">正向效应</text><rect x="146" y="8" width="10" height="10" fill="#10B981"/><text x="160" y="17" font-size="10" fill="#333">负向效应</text><rect x="222" y="8" width="10" height="10" fill="#F59E0B"/><text x="236" y="17" font-size="10" fill="#333">零效应</text><text x="20" y="290" font-size="11" fill="#333333" font-style="italic">图 1. 各结果类型的正向 / 负向 / 零效应证据条数(基于 effect_direction;出版级学术图,不随主题变化)。来源:EduEvidence result.json。</text></svg><figcaption>图 1. 各结果类型的正向 / 负向 / 零效应证据数量(基于 effect_direction,不等同于 Claim 是否被支持)。</figcaption></figure><div class='matrix-controls'><div class='matrix-tools'><input id='matrix-search-full-zh' type='search' placeholder='搜索证据…' aria-label='筛选 / 搜索证据'><select id='matrix-direction-full-zh' aria-label='按效应方向筛选'><option value=''>全部效应</option><option value='positive'>正向效应</option><option value='negative'>负向效应</option><option value='null'>零效应</option></select><select id='matrix-outcome-full-zh' aria-label='按结果类型筛选'><option value=''>全部结果</option><option value='assignment_score'>作业成绩</option><option value='code_quality'>代码质量</option><option value='completion_time'>完成时间</option><option value='independent_problem_solving'>独立问题解决</option><option value='knowledge_gain'>知识获得</option><option value='metacognition'>元认知</option><option value='over_reliance'>过度依赖</option><option value='retention'>记忆保持</option></select></div></div><div class='table-wrap matrix-wrap'><table id='evidence-matrix-full-zh' class='data-table evidence-matrix'><thead><tr><th>ID</th><th>结果</th><th>效应</th><th>质量</th><th>主张</th><th>来源</th></tr></thead><tbody><tr data-effect="positive" data-direction="positive" data-outcome="completion_time" data-search="e-001 study-kazemitabaar-2023 smpl-kazemitabaar-2023-n69 studying the effect of ai code generators on supporting novice learners in introductory programming ai 编程助手在训练期间显著提升任务完成速度与完成率(完成率 1.15 倍、用时 0.57 倍、正确率 1.8 倍)。 69 名 10-17 岁编程新手,此前无文本编程经验 三臂 rct:gpt base(无护栏标准 chatgpt 式界面)与 gpt tutor(护栏版,教师设计提示、不给直接答案)用于数学练习 对照组(无 ai 传统教学) positive s-2023-kazemitabaar"><td><code>E-001</code></td><td><strong>完成时间</strong></td><td><span class="dir pos">正向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">9</span><span class="quality-meter" aria-hidden="true"><i style="width:90%"></i></span></div></td><td class="claim-cell"><p>AI 编程助手在训练期间显著提升任务完成速度与完成率(完成率 1.15 倍、用时 0.57 倍、正确率 1.8 倍)。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-KAZEMITABAAR-2023</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-KAZEMITABAAR-2023-N69</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>随机对照试验</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>K-12 Ages 10 17</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>69 名 10-17 岁编程新手,此前无文本编程经验</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>69</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>三臂 RCT:GPT Base(无护栏标准 ChatGPT 式界面)与 GPT Tutor(护栏版,教师设计提示、不给直接答案)用于数学练习</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>对照组(无 AI 传统教学)</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>Code Authoring Task Progress And Time</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>1.15x completion rate, 0.57x time, 1.8x correctness</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>正向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>3 Weeks Training</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>controlled experiment with random assignment, immediate post-test and 1-week retention test</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>randomized_controlled_design;immediate_post_test_and_retention_test;code_modification_task_guard</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>non_university_population_ages_10_17;small_sample_69;self-paced environment differs from classroom</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>Prior Programming Competency Interaction</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=2 · D2 样本质量=2 · D3 测量效度=2 · D4 时间强度=2 · D5 直接性=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>9.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=partial_novice_programmers_but_younger · subject_match=introductory_programming · tool_match=codex_like_generative_ai · scope=task_performance_during_training</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.7</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>AI 编程助手在训练期间显著提升任务完成速度与完成率(完成率 1.15 倍、用时 0.57 倍、正确率 1.8 倍)。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3544548.3580919</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">https://dl.acm.org/doi/10.1145/3544548.3580919</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3544548.3580919" title="Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming"><code>S-2023-kazemitabaar</code></a></td></tr><tr data-effect="null" data-direction="null" data-outcome="independent_problem_solving" data-search="e-002 study-kazemitabaar-2023 smpl-kazemitabaar-2023-n69 studying the effect of ai code generators on supporting novice learners in introductory programming 可使用 ai 生成代码并未降低学生在人工代码修改任务上的表现(组间差异不显著)。 69 名 10-17 岁编程新手,此前无文本编程经验 无护栏 gpt base(类标准 chatgpt 界面)课内练习;移除访问后参加独立考试 对照组(从未使用 ai) null s-2023-kazemitabaar"><td><code>E-002</code></td><td><strong>独立问题解决</strong></td><td><span class="dir neu">零效应</span><span class="relation-note">与主张关系: 中性</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>可使用 AI 生成代码并未降低学生在人工代码修改任务上的表现(组间差异不显著)。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-KAZEMITABAAR-2023</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-KAZEMITABAAR-2023-N69</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>随机对照试验</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>K-12 Ages 10 17</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>69 名 10-17 岁编程新手,此前无文本编程经验</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>69</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>无护栏 GPT Base(类标准 ChatGPT 界面)课内练习;移除访问后参加独立考试</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>对照组(从未使用 AI)</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>manual code-modification tasks during training</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>no significant difference between groups</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>零效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>中性</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>3 Weeks Training</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>controlled experiment, code-modification task followed each code-authoring task</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>direct_test_of_transfer-adjacent_skill;same_session_measurement</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>code modification is not full independent problem solving;non_university population</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>Practice Effect</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=2 · D2 样本质量=2 · D3 测量效度=1 · D4 时间强度=1 · D5 直接性=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=partial · subject_match=introductory_programming · tool_match=codex · scope=short-term manual code modification</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.5</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>可使用 AI 生成代码并未降低学生在人工代码修改任务上的表现(组间差异不显著)。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3544548.3580919</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">https://dl.acm.org/doi/10.1145/3544548.3580919</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3544548.3580919" title="Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming"><code>S-2023-kazemitabaar</code></a></td></tr><tr data-effect="null" data-direction="null" data-outcome="retention" data-search="e-003 study-kazemitabaar-2023 smpl-kazemitabaar-2023-n69 studying the effect of ai code generators on supporting novice learners in introductory programming 训练结束一周后,codex 组与基线组的保持性差异未达统计显著(codex 组略优)。 69 名 10-17 岁编程新手,此前无文本编程经验 gpt tutor(护栏版:教师设计提示、不给直接答案)用于数学练习 对照组(无 ai)与 gpt base(无护栏)组 null s-2023-kazemitabaar"><td><code>E-003</code></td><td><strong>记忆保持</strong></td><td><span class="dir neu">零效应</span><span class="relation-note">与主张关系: 中性</span></td><td><div class="quality-cell"><span class="num">9</span><span class="quality-meter" aria-hidden="true"><i style="width:90%"></i></span></div></td><td class="claim-cell"><p>训练结束一周后,Codex 组与基线组的保持性差异未达统计显著(Codex 组略优)。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-KAZEMITABAAR-2023</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-KAZEMITABAAR-2023-N69</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>随机对照试验</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>K-12 Ages 10 17</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>69 名 10-17 岁编程新手,此前无文本编程经验</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>69</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>GPT Tutor(护栏版:教师设计提示、不给直接答案)用于数学练习</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>对照组(无 AI)与 GPT Base(无护栏)组</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>retention post-test one week after training</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>slightly better for Codex group but not significant</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>零效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>中性</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>3 Weeks Training Plus 1 Week Retention</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>controlled experiment with delayed retention test</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>delayed_test_included</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>1-week retention window is short;small sample;non-university population</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>Prior Competency</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=2 · D2 样本质量=2 · D3 测量效度=2 · D4 时间强度=2 · D5 直接性=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>9.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=partial · subject_match=introductory_programming · tool_match=codex · scope=retention over one week</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.5</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>训练结束一周后,Codex 组与基线组的保持性差异未达统计显著(Codex 组略优)。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3544548.3580919</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">https://dl.acm.org/doi/10.1145/3544548.3580919</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3544548.3580919" title="Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming"><code>S-2023-kazemitabaar</code></a></td></tr><tr data-effect="negative" data-direction="negative" data-outcome="independent_problem_solving" data-search="e-004 study-bastani-2025 smpl-bastani-2025-n950 generative ai without guardrails can harm learning: evidence from high school mathematics 无护栏使用 gpt-4 的学生在练习表现更高的同时,独立考试成绩比对照组低 17%。 土耳其近千名高中数学学生(约 1000 名学生,共 2848 次观测) 训练阶段一半学习者可使用 openai codex 完成代码编写任务,任务后接代码修改任务 无 codex 访问组 negative s-2025-bastani"><td><code>E-004</code></td><td><strong>独立问题解决</strong></td><td><span class="dir neg">负向效应</span><span class="relation-note">与主张关系: 反驳</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>无护栏使用 GPT-4 的学生在练习表现更高的同时,独立考试成绩比对照组低 17%。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-BASTANI-2025</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-BASTANI-2025-N950</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>随机对照试验</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>高中</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>土耳其近千名高中数学学生(约 1000 名学生,共 2848 次观测)</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>950</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>训练阶段一半学习者可使用 OpenAI Codex 完成代码编写任务,任务后接代码修改任务</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>无 Codex 访问组</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>exam without access to AI resources after practice phase</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>negative_17_percent_on_independent_exam</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>负向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>反驳</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>In Class Study Sessions</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>large-scale randomized controlled trial, practice phase then closed-book exam</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>large_scale_rct;independent_exam_without_ai;arm_wise_design</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>high_school_mathematics_not_university_programming;single_country</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>Tool Design Difference</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=2 · D2 样本质量=2 · D3 测量效度=2 · D4 时间强度=1 · D5 直接性=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=partial_same_age_band_different_subject · subject_match=no_mathematics_vs_programming · tool_match=gpt4_chat_interface · scope=unguarded_general_chat_interface</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.75</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>无护栏使用 GPT-4 的学生在练习表现更高的同时,独立考试成绩比对照组低 17%。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://www.pnas.org/doi/10.1073/pnas.2422633122</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">https://www.pnas.org/doi/10.1073/pnas.2422633122</a></dd></div></dl></details></td><td><a class="source-link" href="https://www.pnas.org/doi/10.1073/pnas.2422633122" title="Generative AI without guardrails can harm learning: Evidence from high school mathematics"><code>S-2025-bastani</code></a></td></tr><tr data-effect="null" data-direction="null" data-outcome="independent_problem_solving" data-search="e-005 study-bastani-2025 smpl-bastani-2025-n950 generative ai without guardrails can harm learning: evidence from high school mathematics ai 导师的护栏设计(给提示而非直接答案、教师参与设计提问)基本消除了负向学习效应,但未观察到正向效应。 土耳其近千名高中数学学生(约 1000 名学生,共 2848 次观测) 训练阶段使用 openai codex 完成代码编写任务 无 codex 访问组;一周后评估后测 null s-2025-bastani"><td><code>E-005</code></td><td><strong>独立问题解决</strong></td><td><span class="dir neu">零效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>AI 导师的护栏设计(给提示而非直接答案、教师参与设计提问)基本消除了负向学习效应,但未观察到正向效应。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-BASTANI-2025</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-BASTANI-2025-N950</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>随机对照试验</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>高中</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>土耳其近千名高中数学学生(约 1000 名学生,共 2848 次观测)</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>950</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>训练阶段使用 OpenAI Codex 完成代码编写任务</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>无 Codex 访问组;一周后评估后测</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>exam without access to AI resources after practice phase</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>negative effect essentially eradicated, no positive effect observed</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>零效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>In Class Study Sessions</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>large-scale randomized controlled trial, three arms</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>direct_manipulation_of_tool_design;large_sample</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>no_positive_learning_gain_even_with_guardrails;subject_mismatch</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>Prompt Engineering Effort</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=2 · D2 样本质量=2 · D3 测量效度=2 · D4 时间强度=1 · D5 直接性=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=partial · subject_match=no · tool_match=guardrailed_tutor_design · scope=guardrail_design_principle_transferable</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.75</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>AI 导师的护栏设计(给提示而非直接答案、教师参与设计提问)基本消除了负向学习效应,但未观察到正向效应。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://www.pnas.org/doi/10.1073/pnas.2422633122</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">https://www.pnas.org/doi/10.1073/pnas.2422633122</a></dd></div></dl></details></td><td><a class="source-link" href="https://www.pnas.org/doi/10.1073/pnas.2422633122" title="Generative AI without guardrails can harm learning: Evidence from high school mathematics"><code>S-2025-bastani</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="assignment_score" data-search="e-006 study-bastani-2025 smpl-bastani-2025-n950 generative ai without guardrails can harm learning: evidence from high school mathematics 练习阶段使用 gpt-4 提升了任务表现(gpt base 组 +48%、gpt tutor 组 +127%),但该任务表现并未迁移到独立考试。 土耳其近千名高中数学学生(约 1000 名学生,共 2848 次观测) 学生在学术写作过程中使用 chatgpt 的体验与策略(质性研究,无效应量测量) 无对照组(质性案例研究) positive s-2025-bastani"><td><code>E-006</code></td><td><strong>作业成绩</strong></td><td><span class="dir pos">正向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>练习阶段使用 GPT-4 提升了任务表现(GPT Base 组 +48%、GPT Tutor 组 +127%),但该任务表现并未迁移到独立考试。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-BASTANI-2025</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-BASTANI-2025-N950</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>随机对照试验</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>高中</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>土耳其近千名高中数学学生(约 1000 名学生,共 2848 次观测)</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>950</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>学生在学术写作过程中使用 ChatGPT 的体验与策略(质性研究,无效应量测量)</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>无对照组(质性案例研究)</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>practice problem performance during study sessions</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>48-127 percent improvement on practice problems</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>正向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>In Class Study Sessions</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>randomized controlled trial with practice and closed-book exam phases</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>same_study_compares_task_and_learning;large_sample</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>subject_mismatch_mathematics</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>Task Familiarity</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=2 · D2 样本质量=2 · D3 测量效度=2 · D4 时间强度=1 · D5 直接性=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=partial · subject_match=no · tool_match=gpt4 · scope=task_performance_vs_learning_separation</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.75</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>练习阶段使用 GPT-4 提升了任务表现(GPT Base 组 +48%、GPT Tutor 组 +127%),但该任务表现并未迁移到独立考试。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://www.pnas.org/doi/10.1073/pnas.2422633122</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">https://www.pnas.org/doi/10.1073/pnas.2422633122</a></dd></div></dl></details></td><td><a class="source-link" href="https://www.pnas.org/doi/10.1073/pnas.2422633122" title="Generative AI without guardrails can harm learning: Evidence from high school mathematics"><code>S-2025-bastani</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="knowledge_gain" data-search="e-007 study-marzuki-2024 smpl-marzuki-2024-n72 impact of chatgpt on esl students&#x27; academic writing skills 以 chatgpt 作为形成性反馈工具,对学生学术写作能力产生了显著的正向影响,学生评价亦为正面。 印度某大学本科英语作为第二语言(esl)学生,n=72 学生在学术写作过程中使用 chatgpt 的体验与策略 无对照组(质性案例研究) positive s-2024-marzuki"><td><code>E-007</code></td><td><strong>知识获得</strong></td><td><span class="dir pos">正向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">6</span><span class="quality-meter" aria-hidden="true"><i style="width:60%"></i></span></div></td><td class="claim-cell"><p>以 ChatGPT 作为形成性反馈工具,对学生学术写作能力产生了显著的正向影响,学生评价亦为正面。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-MARZUKI-2024</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-MARZUKI-2024-N72</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Impact of ChatGPT on ESL students&#x27; academic writing skills</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Impact of ChatGPT on ESL students&#x27; academic writing skills</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2024</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>混合方法</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>Undergraduate</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>印度某大学本科英语作为第二语言(ESL)学生,n=72</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>72</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>学生在学术写作过程中使用 ChatGPT 的体验与策略</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>无对照组(质性案例研究)</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>writing tests with pre-post-delayed design</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>significant positive impact on writing skills</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>正向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>6 Hours Intervention</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>mixed methods intervention study, pre/post/delayed tests and focus groups</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>delayed_post_test;mixed_methods_triangulation</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>short_intervention_6_hours;single_institution;elite_private_university</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>Self Selection Consent</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=1 · D2 样本质量=1 · D3 测量效度=2 · D4 时间强度=2 · D5 直接性=0</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>6.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=yes_undergraduate · subject_match=no_writing_not_programming · tool_match=chatgpt · scope=formative_feedback_writing</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>以 ChatGPT 作为形成性反馈工具,对学生学术写作能力产生了显著的正向影响,学生评价亦为正面。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://link.springer.com/article/10.1186/s40561-024-00295-9</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://link.springer.com/article/10.1186/s40561-024-00295-9">https://link.springer.com/article/10.1186/s40561-024-00295-9</a></dd></div></dl></details></td><td><a class="source-link" href="https://link.springer.com/article/10.1186/s40561-024-00295-9" title="Impact of ChatGPT on ESL students&#x27; academic writing skills"><code>S-2024-marzuki</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="completion_time" data-search="e-008 study-peng-2023 smpl-peng-2023-n95 the impact of ai on developer productivity: evidence from github copilot 随机对照实验(n=95)显示:使用 copilot 的职业开发者完成标准化编码任务的用时比对照组缩短约 55%。 95 名经自由职业平台招募的职业开发者,完成标准化编码任务 任务期间可使用 github copilot 不可使用 copilot 的对照组 positive s-2023-peng"><td><code>E-008</code></td><td><strong>完成时间</strong></td><td><span class="dir pos">正向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>随机对照实验(n=95)显示:使用 Copilot 的职业开发者完成标准化编码任务的用时比对照组缩短约 55%。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-PENG-2023</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-PENG-2023-N95</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>The Impact of AI on Developer Productivity: Evidence from GitHub Copilot</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>The Impact of AI on Developer Productivity: Evidence from GitHub Copilot</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>随机对照试验</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>Professional Developers Not Students</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>95 名经自由职业平台招募的职业开发者,完成标准化编码任务</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>95</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>任务期间可使用 GitHub Copilot</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>不可使用 Copilot 的对照组</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>Time To Complete Http Server Implementation</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>~55.8% faster task completion in Copilot group</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>正向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>Single Task Session</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>online randomized controlled experiment with objective completion-time metric</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>randomized_controlled_design;objective_completion_time_metric</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>professional_population_not_students;single_task_ecology;preprint_not_peer_reviewed</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>Task Familiarity;Platform Recruitment Self Selection</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=2 · D2 样本质量=2 · D3 测量效度=2 · D4 时间强度=1 · D5 直接性=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=mismatch_professional_developers · subject_match=adjacent_web_development_task · tool_match=copilot_like_generative_ai · scope=task_performance_only_no_learning_outcome</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.6</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>随机对照实验(n=95)显示:使用 Copilot 的职业开发者完成标准化编码任务的用时比对照组缩短约 55%。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://doi.org/10.48550/arXiv.2302.06590</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://doi.org/10.48550/arXiv.2302.06590">https://doi.org/10.48550/arXiv.2302.06590</a></dd></div></dl></details></td><td><a class="source-link" href="https://doi.org/10.48550/arXiv.2302.06590" title="The Impact of AI on Developer Productivity: Evidence from GitHub Copilot"><code>S-2023-peng</code></a></td></tr><tr data-effect="null" data-direction="null" data-outcome="code_quality" data-search="e-009 study-yetistiren-2023 smpl-yetistiren-2023-bench github copilot ai pair programmer: asset or liability? 系统性基准评估显示 copilot 生成代码相对人类代码的质量结论不一:部分基准上正确性具竞争力,同时记录到安全相关缺陷。 取自公开基准数据集的 copilot 生成程序与人类编写程序 copilot 生成的程序 相同基准上的人类编写程序 null s-2023-yetistiren"><td><code>E-009</code></td><td><strong>代码质量</strong></td><td><span class="dir neu">零效应</span><span class="relation-note">与主张关系: 中性</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>系统性基准评估显示 Copilot 生成代码相对人类代码的质量结论不一:部分基准上正确性具竞争力,同时记录到安全相关缺陷。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-YETISTIREN-2023</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-YETISTIREN-2023-BENCH</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>GitHub Copilot AI Pair Programmer: Asset or Liability?</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>GitHub Copilot AI Pair Programmer: Asset or Liability?</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>观察性研究</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>Not Applicable Code Artifacts</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>取自公开基准数据集的 Copilot 生成程序与人类编写程序</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>Copilot 生成的程序</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>相同基准上的人类编写程序</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>Correctness Security Maintainability Metrics On Benchmarks</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>mixed quality profile; no single-direction summary</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>零效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>中性</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>Not Applicable Artifact Study</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>systematic empirical evaluation of generated code against human baselines on public benchmarks</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>multi_dimensional_quality_metrics;reproducible_benchmark_protocol</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>artifact_benchmark_not_classroom;no_learning_outcome;tool_version_from_2023</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>Benchmark Task Distribution</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=1 · D2 样本质量=2 · D3 测量效度=2 · D4 时间强度=1 · D5 直接性=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=mismatch_no_learners_in_study · subject_match=introductory_adjacent_code_tasks · tool_match=copilot_like_generative_ai · scope=output_quality_only</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>系统性基准评估显示 Copilot 生成代码相对人类代码的质量结论不一:部分基准上正确性具竞争力,同时记录到安全相关缺陷。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://doi.org/10.1016/j.jss.2023.111734</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://doi.org/10.1016/j.jss.2023.111734">https://doi.org/10.1016/j.jss.2023.111734</a></dd></div></dl></details></td><td><a class="source-link" href="https://doi.org/10.1016/j.jss.2023.111734" title="GitHub Copilot AI Pair Programmer: Asset or Liability?"><code>S-2023-yetistiren</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="assignment_score" data-search="e-010 study-finnieansley-2022 smpl-finnieansley-2022-qsets using github copilot to solve introductory programming problems codex 在 cs1 考试风格题目上能给出通过水平的解答(依数据集约 50%–75%),表明新手手中存在可观的任务能力余量。 cs1 考试风格题目集,由 codex 作答并与已发表的学生分数分布比较 codex 对 cs1 题目作答生成 已发表的学生同届分数分布 positive s-2022-finnie-ansley"><td><code>E-010</code></td><td><strong>作业成绩</strong></td><td><span class="dir pos">正向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>Codex 在 CS1 考试风格题目上能给出通过水平的解答(依数据集约 50%–75%),表明新手手中存在可观的任务能力余量。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-FINNIEANSLEY-2022</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-FINNIEANSLEY-2022-QSETS</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Using GitHub Copilot to Solve Introductory Programming Problems</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Using GitHub Copilot to Solve Introductory Programming Problems</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2022</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>观察性研究</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>University Year 1 Question Sets</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>CS1 考试风格题目集,由 Codex 作答并与已发表的学生分数分布比较</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>Codex 对 CS1 题目作答生成</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>已发表的学生同届分数分布</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>Pass Rate On CS1 Exam Style Questions</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>passing solutions on ~50-75% of questions across datasets</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>正向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>Not Applicable Capability Probe</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>capability benchmark against published student distributions; reproducible question sets</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>public_reproducible_question_sets;directly_relevant_task_domain</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>tool_solves_task_does_not_equate_student_learning;codex_2021_model_version_outdated</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>Question Leakage Into Training Data Possible</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=1 · D2 样本质量=2 · D3 测量效度=2 · D4 时间强度=1 · D5 直接性=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=partial_measures_tool_not_students · subject_match=introductory_programming · tool_match=copilot_like_generative_ai · scope=tool_capability_headroom</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>Codex 在 CS1 考试风格题目上能给出通过水平的解答(依数据集约 50%–75%),表明新手手中存在可观的任务能力余量。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3545945.3569830</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3545945.3569830">https://dl.acm.org/doi/10.1145/3545945.3569830</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3545945.3569830" title="Using GitHub Copilot to Solve Introductory Programming Problems"><code>S-2022-finnie-ansley</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="metacognition" data-search="e-011 study-explcomp-2023 smpl-explcomp-2023-ratings comparing code explanations created by students and large language models 受控比较发现 llm 生成的代码讲解与学生自撰讲解相当(部分更优),适合作为解释性支架材料,而非替代学生的解释练习。 同一批短程序的学生版与 llm 版讲解的受控对比 llm 生成的代码讲解 学生撰写的同题讲解 positive s-2023-explanations-compare"><td><code>E-011</code></td><td><strong>元认知</strong></td><td><span class="dir pos">正向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>受控比较发现 LLM 生成的代码讲解与学生自撰讲解相当(部分更优),适合作为解释性支架材料,而非替代学生的解释练习。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-EXPLCOMP-2023</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-EXPLCOMP-2023-RATINGS</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Comparing Code Explanations Created by Students and Large Language Models</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Comparing Code Explanations Created by Students and Large Language Models</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>观察性研究</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>University Introductory</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>同一批短程序的学生版与 LLM 版讲解的受控对比</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>LLM 生成的代码讲解</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>学生撰写的同题讲解</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>Rated Explanation Quality And Comprehensibility</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>comparable-or-better rated quality vs student explanations</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>正向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>Single Session Ratings</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>controlled comparison with blind rating of explanation pairs</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>controlled_pairwise_comparison;learning_process_relevant_construct</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>short_term_ratings_not_learning_gains;small_program_snippets_ecology</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>Rating Criteria Subjectivity</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=1 · D2 样本质量=2 · D3 测量效度=2 · D4 时间强度=1 · D5 直接性=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=partial_scaffold_material_only · subject_match=introductory_programming · tool_match=llm_explanations · scope=scaffold_quality_not_effectiveness</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>受控比较发现 LLM 生成的代码讲解与学生自撰讲解相当(部分更优),适合作为解释性支架材料,而非替代学生的解释练习。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3587102.3588785</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3587102.3588785">https://dl.acm.org/doi/10.1145/3587102.3588785</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3587102.3588785" title="Comparing Code Explanations Created by Students and Large Language Models"><code>S-2023-explanations-compare</code></a></td></tr><tr data-effect="negative" data-direction="negative" data-outcome="over_reliance" data-search="e-012 study-vaithilingam-2022 smpl-vaithilingam-2022-n24 expectation vs. experience: evaluating the usability of code generation tools 尽管首任务完成更快,参与者难以理解并调试 ai 生成的解法、对最终程序所有权感低——记录了纯速度指标遗漏的元认知与依赖风险。 24 名参与者参与的 copilot 类工具组内可用性研究 copilot 类工具辅助编程 不使用工具的组内基线 negative s-2022-vaithilingam"><td><code>E-012</code></td><td><strong>过度依赖</strong></td><td><span class="dir neg">负向效应</span><span class="relation-note">与主张关系: 反驳</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>尽管首任务完成更快,参与者难以理解并调试 AI 生成的解法、对最终程序所有权感低——记录了纯速度指标遗漏的元认知与依赖风险。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>STUDY-VAITHILINGAM-2022</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SMPL-VAITHILINGAM-2022-N24</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Expectation vs. Experience: Evaluating the Usability of Code Generation Tools</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Expectation vs. Experience: Evaluating the Usability of Code Generation Tools</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2022</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>质性研究</dd></div><div class="evidence-detail-row"><dt>教育阶段</dt><dd>Mixed Cs Students And Professionals</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>24 名参与者参与的 Copilot 类工具组内可用性研究</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>24</dd></div><div class="evidence-detail-row"><dt>干预</dt><dd>Copilot 类工具辅助编程</dd></div><div class="evidence-detail-row"><dt>对照 / 比较条件</dt><dd>不使用工具的组内基线</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>Understanding Ownership And Debugging Reports</dd></div><div class="evidence-detail-row"><dt>效应 / 结果</dt><dd>documented comprehension/ownership difficulties despite speed gain</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>负向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>反驳</dd></div><div class="evidence-detail-row"><dt>干预时长</dt><dd>Single Session</dd></div><div class="evidence-detail-row"><dt>方法</dt><dd>within-subject usability study with tasks, observation and interviews</dd></div><div class="evidence-detail-row"><dt>优势</dt><dd>rich_qualitative_process_data;constructs_missed_by_speed_metrics</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>small_n_24;single_session;self_reported_understanding</dd></div><div class="evidence-detail-row"><dt>混杂因素</dt><dd>Participant AI Familiarity</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=1 · D2 样本质量=2 · D3 测量效度=2 · D4 时间强度=1 · D5 直接性=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>证据等级</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>适用性</dt><dd>learner_match=partial_includes_cs_students · subject_match=programming_adjacent · tool_match=copilot_like_generative_ai · scope=risk_identification</dd></div><div class="evidence-detail-row"><dt>置信度</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>被反驳</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>尽管首任务完成更快,参与者难以理解并调试 AI 生成的解法、对最终程序所有权感低——记录了纯速度指标遗漏的元认知与依赖风险。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3491101.3519665</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3491101.3519665">https://dl.acm.org/doi/10.1145/3491101.3519665</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3491101.3519665" title="Expectation vs. Experience: Evaluating the Usability of Code Generation Tools"><code>S-2022-vaithilingam</code></a></td></tr></tbody></table></div></div></section><section id="full-03-quality" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>03 证据可信度、反证与方法审计</h2><p class="full-chapter-lead">检查证据为什么可信、哪里冲突,以及哪些结论必须降级。</p></header><div class="full-chapter-body"><section class="method-review-group"><header class="method-review-title"><h3>审查目标:overall</h3><span class="method-verdict">关注</span></header><div class="method-audit-grid"><article class="method-audit-item method-met"><div class="method-audit-head"><strong>对照组</strong><span class="method-status">通过</span></div><p>提出因果主张的量化研究均含对照条件:Bastani(E-004/E-005/E-006)为三臂随机对照(无护栏 GPT Base / 护栏 GPT Tutor / 无 AI 对照,约千名学生);…</p><details class="method-detail detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>提出因果主张的量化研究均含对照条件:Bastani(E-004/E-005/E-006)为三臂随机对照(无护栏 GPT Base / 护栏 GPT Tutor / 无 AI 对照,约千名学生);Kazemitabaar(E-001/E-002/E-003)为随机对照(有/无 Codex,n=69);Peng(E-008)为职业开发者随机对照(n=95)。不含对照的是不承担因果主张的研究:Marzuki(E-007)为混合方法,Vaithilingam(E-012)为组内可用性研究(n=24),Yetistiren(E-009)与Finnie-Ansley(E-010)为基准评估。</p></div></details></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>随机分配</strong><span class="method-status">通过</span></div><p>Kazemitabaar(E-001/E-002/E-003)与 Bastani(E-004/E-005/E-006)均为随机分配。</p></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>前测</strong><span class="method-status">通过</span></div><p>Kazemitabaar 有前测评估;Bastani 测量基线协变量。</p></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>后测</strong><span class="method-status">通过</span></div><p>三项随机对照研究均报告即时后测。</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>保持测试</strong><span class="method-status">部分</span></div><p>Kazemitabaar(E-003)有一周保持测;Bastani(E-004/E-005/E-006)无延迟测验;Marzuki(E-007)有延迟测量。</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>迁移测试</strong><span class="method-status">部分</span></div><p>Kazemitabaar 的代码修改任务属迁移邻近任务;本证据集缺少完整的无 AI 迁移测验。</p></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>样本偏差</strong><span class="method-status">通过</span></div><p>Bastani 样本接近千人;Kazemitabaar 样本偏小且年龄偏低(n=69,10-17 岁)。</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>自我选择偏差</strong><span class="method-status">部分</span></div><p>Marzuki(E-007)基于知情同意招募,存在自我选择风险。</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>测量效度</strong><span class="method-status">部分</span></div><p>完成率/正确率/速度作为&#x27;学习&#x27;代理的效度可疑:E-001 练习正确率、E-004 任务得分、E-011/E-012 完成速度与进展均在 AI 可访问条件下测得,AI 可直接产出答案抬高指标,无法区分&#x27;学会了&#x27;与&#x27;抄到了&#x27;(Bastani 机制数据 E-002:GPT Base 答对率 5…</p><details class="method-detail detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>完成率/正确率/速度作为&#x27;学习&#x27;代理的效度可疑:E-001 练习正确率、E-004 任务得分、E-011/E-012 完成速度与进展均在 AI 可访问条件下测得,AI 可直接产出答案抬高指标,无法区分&#x27;学会了&#x27;与&#x27;抄到了&#x27;(Bastani 机制数据 E-002:GPT Base 答对率 51% 中 42% 为逻辑错误;Wermelinger S-2023 显示 Copilot 可首次尝试解决 24 道典型入门题中的 16 道,FETCH_PARTIAL 仅验证到机构库摘要)。自评测量不可靠(E-002 学生过度乐观、E-008 能力错觉)。效度较高的测量(无 AI 独立考试 E-002/E-003、延迟后测 E-005)恰恰给出负向或零结果——测量选择本身决定结论方向。</p></div></details></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>混杂因素</strong><span class="method-status">部分</span></div><p>Kazemitabaar 中先验编程能力与 AI 收益存在交互。</p></article><article class="method-audit-item method-not_applicable"><div class="method-audit-head"><strong>教师效应</strong><span class="method-status">不适用</span></div><p>Kazemitabaar 为自定进度;课堂类研究可能带有教师效应。</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>新奇效应</strong><span class="method-status">部分</span></div><p>短周期干预易高估参与度;本证据集的研究均未控制新奇效应。</p></article><article class="method-audit-item method-not_applicable"><div class="method-audit-head"><strong>工具版本效应</strong><span class="method-status">不适用</span></div><p>各研究只覆盖单一工具版本,工具迭代快,结论耐久性受限。</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>AI 使用规则</strong><span class="method-status">部分</span></div><p>Bastani(E-001~E-003)是唯一直接操纵 AI 使用政策的研究:无护栏 GPT Base(类标准 ChatGPT 界面,可抄答案)vs 护栏 GPT Tutor(教师设计提示、不给直接答案),对应实证了&#x27;允许使用但无规则→独立考试 -17% 伤害&#x27;与&#x27;有护栏→练习 +127%…</p><details class="method-detail detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>Bastani(E-001~E-003)是唯一直接操纵 AI 使用政策的研究:无护栏 GPT Base(类标准 ChatGPT 界面,可抄答案)vs 护栏 GPT Tutor(教师设计提示、不给直接答案),对应实证了&#x27;允许使用但无规则→独立考试 -17% 伤害&#x27;与&#x27;有护栏→练习 +127% 且负效应消除&#x27;的政策对比,与试点&#x27;禁止直接提交 AI 代码、实验课独立评测&#x27;的护栏设计同构。局限:护栏效果仅在单一情境(高中数学、教师设计提示)验证过,需在 C 语言场景复验;且各研究均未验证对照组依从性(对照组成员是否实际未使用 AI),政策污染未排除。</p></div></details></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>样本流失</strong><span class="method-status">部分</span></div><p>Marzuki(E-007)报告了流失情况;其余研究未详细说明。</p></article></div><div class="method-guard-wrap"><strong>任务 vs 学习护栏:</strong><div class="method-guard"><p>证据集本身未把任务表现等同学习效果:三类独立测量(E-002/E-003/E-005)与任务表现测量(E-001/E-004/E-011/E-012)被明确分开,且独立测量给出负向/零结果,恰是对照铁律(SKILL.md RULE 3:task performance 不得自动等同 learning effect)的正确执行。…</p><details class="detail-expander"><summary>查看审计依据</summary><div class="detail-body"><p>证据集本身未把任务表现等同学习效果:三类独立测量(E-002/E-003/E-005)与任务表现测量(E-001/E-004/E-011/E-012)被明确分开,且独立测量给出负向/零结果,恰是对照铁律(SKILL.md RULE 3:task performance 不得自动等同 learning effect)的正确执行。但风险在边界处:(a) 若下游综合以 E-001 的 +127% 或 E-011 的快 35% 作为&#x27;学习提升&#x27;证据,即违反铁律,证据天平会系统性偏向&#x27;允许使用&#x27;;(b) E-001 练习成绩提升与 E-002 独立考试伤害在同一研究中并存,任何只引其一的做法都会误导。frame 的 outcomes 已把&#x27;任务表现&#x27;与&#x27;学习能力&#x27;分开测量、并把期末无 AI 统一机试/笔试设为唯一成功判据,与本 guard 一致。</p></div></details></div></div></section><article class="conflict-card"><strong>裁决说明:</strong><p class="conflict-text">分歧来自结果分离(任务 vs 学习)、工具设计(有护栏 vs 无护栏)与人群(K-12/职业者 vs 大学生)。随机实验与基准研究中任务表现证据一致为正;唯一测量移除 AI 后独立表现的研究显示无护栏时有害;可用性与工件研究补充依赖与质量警示而非解决学习问题。</p></article><div class="evidence-tribunal" data-visual="evidence-tribunal-grid"><p class="tribunal-summary"><strong>决策:</strong>试点验证 · <strong>置信度:</strong>中</p><div class="tribunal-grid"><article class="tribunal-card supported"><header><span aria-hidden="true">✓</span><h3>可以主张</h3><span class="tribunal-count">7</span></header><ul><li><p>AI 编程助手在训练期间可靠地提升任务表现(完成速度、正确性)—— E-001、E-006。</p><div class="tribunal-evidence-refs"><code>E-001</code><code>E-006</code></div></li><li><p>无护栏的生成式 AI 访问在移除工具后可能损害独立问题解决能力 —— E-004。</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li><li><p>护栏设计(给提示而非给答案)能大幅缓解负面学习效应 —— E-005。</p><div class="tribunal-evidence-refs"><code>E-005</code></div></li><li><p>任务表现提升并不自动等于学习提升 —— E-004 与 E-006 的研究内对照。</p><div class="tribunal-evidence-refs"><code>E-004</code><code>E-006</code></div></li><li><p>工具能力可观:Codex 能解出约半数至四分之三的 CS1 考试风格题目 —— E-010。</p><div class="tribunal-evidence-refs"><code>E-010</code></div></li><li><p>职业开发者 RCT 显示 Copilot 带来约 55% 任务提速;但职业人群限制直接性 —— E-008。</p><div class="tribunal-evidence-refs"><code>E-008</code></div></li><li><p>LLM 代码讲解的质量评级与学生自撰讲解相当,可作支架材料 —— E-011。</p><div class="tribunal-evidence-refs"><code>E-011</code></div></li></ul></article><article class="tribunal-card uncertain"><header><span aria-hidden="true">?</span><h3>尚不能主张</h3><span class="tribunal-count">4</span></header><ul><li><p>AI 编程助手能否真正改善或保持大学新手的编程学习——本证据集中没有大学层面的直接 RCT [无直接证据]</p></li><li><p>Kazemitabaar 2023 的一周中性保持性能否延伸到一个学期 —— E-003。</p><div class="tribunal-evidence-refs"><code>E-003</code></div></li><li><p>基准质量结论(E-009)与讲解质量评级(E-011)能否转化为课堂学习收益。</p><div class="tribunal-evidence-refs"><code>E-009</code><code>E-011</code></div></li><li><p>可用性研究所记录的理解/所有权困难(E-012)在整学期护栏条件下会如何演变。</p><div class="tribunal-evidence-refs"><code>E-012</code></div></li></ul></article><article class="tribunal-card contradicted"><header><span aria-hidden="true">×</span><h3>被反驳的主张</h3><span class="tribunal-count">2</span></header><ul><li><p>&#x27;AI 工具总能提高学习&#x27;被 E-004 反驳(无护栏访问,独立考试 −17%)。</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li><li><p>&#x27;速度收益等于学习收益&#x27;被 E-001/E-006/E-008 与 E-004 之间的任务-学习分离所反驳。</p><div class="tribunal-evidence-refs"><code>E-001</code><code>E-006</code><code>E-008</code><code>E-004</code></div></li></ul></article><article class="tribunal-card missing"><header><span aria-hidden="true">…</span><h3>缺失证据</h3><span class="tribunal-count">4</span></header><ul><li><p>在大学编程课程中带保持与无 AI 迁移测试的 RCT。</p></li><li><p>同一课程内变化 AI 使用政策的研究。</p></li><li><p>跨越一门课的 AI 依赖纵向数据。</p></li><li><p>职业提速 RCT 的同行评审重复(Peng 等仍为预印本)。</p></li></ul></article></div><details class="supporting-visual"><summary>EvidenceFlow 协议</summary><svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="EvidenceFlow 协议流程"><title>EvidenceFlow 协议流程</title><desc>从问题框架、检索、抓取验证、证据抽取、反方质疑、方法审计、裁决到适用性与干预评价的完整流程。</desc><defs><marker id="arr-zh-full-workflow" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">EvidenceFlow 协议</text><rect x="24" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="56.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="56.0" y="126.0">问题框架</tspan></text><line x1="88" y1="126" x2="100" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="100" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="132.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="132.0" y="126.0">检索</tspan></text><line x1="164" y1="126" x2="176" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="176" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="208.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="208.0" y="126.0">抓取</tspan><tspan x="208.0" y="140.0">验证</tspan></text><line x1="240" y1="126" x2="252" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="252" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="284.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="284.0" y="126.0">证据抽取</tspan></text><line x1="316" y1="126" x2="328" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="328" y="100" width="64" height="52" rx="8" fill="#5E8A6A"/><text x="360.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="360.0" y="126.0">反方质疑</tspan></text><line x1="392" y1="126" x2="404" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="404" y="100" width="64" height="52" rx="8" fill="#5E8A6A"/><text x="436.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="436.0" y="126.0">方法</tspan><tspan x="436.0" y="140.0">审计</tspan></text><line x1="468" y1="126" x2="480" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="480" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="512.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="512.0" y="126.0">裁决</tspan></text><line x1="544" y1="126" x2="556" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="556" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="588.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="588.0" y="126.0">适用性</tspan></text><line x1="620" y1="126" x2="632" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="632" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="664.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="664.0" y="126.0">干预</tspan><tspan x="664.0" y="140.0">评价</tspan></text></svg></details><details class="supporting-visual"><summary>裁决信息图</summary><svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="证据裁决信息图"><title>证据裁决信息图</title><desc>可以主张与不可主张的证据 ID 与建议决策徽章;完整主张文本见下方裁决卡片。</desc><defs><marker id="arr-zh-full-tribunal" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">证据裁决</text><text x="24" y="180" font-size="13" font-weight="700" fill="#3A3833">冲突来源</text><text x="24" y="202" font-size="11" fill="#8A867E">详见下方裁决卡片</text><rect x="24" y="216" width="180" height="28" rx="14" fill="#C99A4A"/><text x="114" y="235" text-anchor="middle" font-size="13" font-weight="700" fill="#fff">试点验证</text><text x="250" y="80" font-size="13" font-weight="700" fill="#5E8A6A">可以主张 (7)</text><circle cx="254" cy="100" r="3" fill="#5E8A6A"/><text x="266" y="105" font-size="11" fill="#3A3833">E-001</text><circle cx="254" cy="122" r="3" fill="#5E8A6A"/><text x="266" y="127" font-size="11" fill="#3A3833">E-006</text><circle cx="254" cy="144" r="3" fill="#5E8A6A"/><text x="266" y="149" font-size="11" fill="#3A3833">E-004</text><circle cx="254" cy="166" r="3" fill="#5E8A6A"/><text x="266" y="171" font-size="11" fill="#3A3833">E-005</text><text x="490" y="80" font-size="13" font-weight="700" fill="#A85B53">不可主张 (2)</text><circle cx="494" cy="100" r="3" fill="#A85B53"/><text x="506" y="105" font-size="11" fill="#3A3833">E-004</text><circle cx="494" cy="122" r="3" fill="#A85B53"/><text x="506" y="127" font-size="11" fill="#3A3833">E-001</text><circle cx="494" cy="144" r="3" fill="#A85B53"/><text x="506" y="149" font-size="11" fill="#3A3833">E-006</text><circle cx="494" cy="166" r="3" fill="#A85B53"/><text x="506" y="171" font-size="11" fill="#3A3833">E-008</text></svg></details></div><div class="trace-chain" data-visual="trace-chain"><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-001</strong><p class="trace-claim-text">在练习环节,无护栏的 GPT Base(类标准 ChatGPT 界面)使高中生的练习成绩相对对照组提高 48%,带护栏的 GPT Tutor 提高 127%(Table 1:practice 系数 0.137/0.361,对照均值 0.284)</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-001</code><span>完成时间</span><span class="dir pos">正向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 9.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3544548.3580919"><code>S-2023-kazemitabaar</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-002</strong><p class="trace-claim-text">移除 AI 访问后的无辅助独立考试中,GPT Base 组成绩比从未使用 AI 的对照组低 17%(统计显著),表明无护栏使用 AI 损害技能习得;机制上学生把 GPT 当&#x27;拐杖&#x27;直接抄答案(GPT Base 答对率仅 51%,其中 42% 逻辑错误、8% 算术错误),且学生自评过度乐观、未察觉学习受损</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-002</code><span>独立问题解决</span><span class="dir neu">零效应</span><span class="trace-relation">主张关系: 中性</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3544548.3580919"><code>S-2023-kazemitabaar</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-003</strong><p class="trace-claim-text">带护栏的 GPT Tutor(教师设计提示而非直接答案)在练习成绩 +127% 的同时,移除访问后的独立考试负效应基本消除(-0.004,不显著),说明精心设计的护栏可兼得练习提升与学习保持</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-003</code><span>记忆保持</span><span class="dir neu">零效应</span><span class="trace-relation">主张关系: 中性</span><span>Q 9.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3544548.3580919"><code>S-2023-kazemitabaar</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-004</strong><p class="trace-claim-text">训练阶段使用 OpenAI Codex 的 10-17 岁新手在 45 道 Python 代码编写任务上表现显著提升:完成率提高 1.15 倍、得分提高 1.8 倍</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-004</code><span>独立问题解决</span><span class="dir neg">负向效应</span><span class="trace-relation">主张关系: 反驳</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122"><code>S-2025-bastani</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-005</strong><p class="trace-claim-text">训练期使用 Codex 的学习者一周后评估后测成绩略好于对照组,但差异未达统计显著(保持力无显著差异);Scratch 前测高分者若有 Codex 使用史,保持后测显著更好</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-005</code><span>独立问题解决</span><span class="dir neu">零效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122"><code>S-2025-bastani</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-006</strong><p class="trace-claim-text">质性案例研究中,3 名 EFL 学生珍视 ChatGPT 的辅助价值(消除不确定性、澄清词汇、提供内容建议、语法/结构反馈,让学生专注于创意层面),并形成语言精修、观点生成与结构、校对与信心增强等使用策略</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-006</code><span>作业成绩</span><span class="dir pos">正向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122"><code>S-2025-bastani</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-007</strong><p class="trace-claim-text">同一质性研究中,学生担忧 AI 使用的学术真实性与过度依赖风险(建议过于复杂/正式、语气不符、文化刻板印象等局限),强调必须保持人的判断并寻求教师/同伴反馈,呼吁伦理指引与批判性思维培养</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-007</code><span>知识获得</span><span class="dir pos">正向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 6.0</span><span class="trace-arrow">→</span><a href="https://link.springer.com/article/10.1186/s40561-024-00295-9"><code>S-2024-marzuki</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-008</strong><p class="trace-claim-text">随机对照实验(n=95)显示:使用 Copilot 的职业开发者完成标准化编程任务的用时比对照组缩短约 55%。</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-008</code><span>完成时间</span><span class="dir pos">正向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://doi.org/10.48550/arXiv.2302.06590"><code>S-2023-peng</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-009</strong><p class="trace-claim-text">系统性基准评估显示 Copilot 生成代码相对人类代码的质量结论不一:部分正确性具竞争力,同时记录到安全相关缺陷。</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-009</code><span>代码质量</span><span class="dir neu">零效应</span><span class="trace-relation">主张关系: 中性</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://doi.org/10.1016/j.jss.2023.111734"><code>S-2023-yetistiren</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-010</strong><p class="trace-claim-text">Codex 在 CS1 考试风格题目上能给出通过水平的解答(依数据集约 50%–75%),表明新手手中存在可观的任务能力余量。</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-010</code><span>作业成绩</span><span class="dir pos">正向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3545945.3569830"><code>S-2022-finnie-ansley</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-011</strong><p class="trace-claim-text">受控比较发现 LLM 生成的代码讲解与学生自撰讲解相当(部分更优),适合作为解释性支架材料,而非替代学生的解释练习。</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-011</code><span>元认知</span><span class="dir pos">正向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3587102.3588785"><code>S-2023-explanations-compare</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-012</strong><p class="trace-claim-text">尽管首任务完成更快,参与者难以理解并调试 AI 生成的解法、对最终程序所有权感低——记录了纯速度指标遗漏的认知与依赖风险。</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-012</code><span>过度依赖</span><span class="dir neg">负向效应</span><span class="trace-relation">主张关系: 反驳</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3491101.3519665"><code>S-2022-vaithilingam</code></a></div></div></article></div><div id="chart-trace-zh" class="chart-mount" aria-label="主张-证据追溯"></div><p class="chart-interpretation"><strong>这意味着什么:</strong>每个重要主张都必须能追到 Evidence ID 和原始来源。</p></div></section><section id="full-04-action" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>04 适用范围与教学行动</h2><p class="full-chapter-lead">把可外推范围、护栏和教学动作连接到具体证据。</p></header><div class="full-chapter-body"><p><strong>目标人群:</strong>首次学习 C 语言编程的大一计算机专业学生</p>
2028
2188
  <p><strong>目标情境:</strong>16 周讲授课+实验课,60 人班级,助教支持,线下</p>
2029
2189
  <p><strong>适用于谁:</strong>在大一 C 课程以护栏化使用政策开展试点</p>
2030
2190
  <p><strong>不适用于:</strong>无使用政策的全面放开采用</p>
@@ -2059,7 +2219,7 @@ select:focus-visible,
2059
2219
  <p><strong>成功阈值:</strong>实验班独立问题解决非劣(差异在 5% 以内)且保持相当或更优、AI 依赖指数低于阈值;若独立问题解决下滑超过 10%,无论任务收益如何,试点均判为失败。</p>
2060
2220
  <p><strong>分析计划:</strong>预登记的实验班 vs 对照班基线调整学习指标比较(ANCOVA);任务表现指标与学习指标分开报告;按先验编程能力做亚组分析;第 3、5、7 周监测停止条件。</p>
2061
2221
  <h3>评价设计信息图</h3>
2062
- <svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="评价设计流程"><title>评价设计流程</title><desc>基线、后测、保持测试与迁移测试的评价流程;完整指标与分析计划见评估章节。</desc><defs><marker id="arr-zh-full-evaluation" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">评价设计流程</text><rect x="30" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="105.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="105.0" y="138.0">基线</tspan></text><text x="105.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">前测</text><line x1="180" y1="138" x2="192" y2="138" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-evaluation)"/><rect x="192" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="267.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="267.0" y="138.0">后测</tspan></text><text x="267.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">后测</text><line x1="342" y1="138" x2="354" y2="138" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-evaluation)"/><rect x="354" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="429.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="429.0" y="138.0">保持</tspan></text><text x="429.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">保持测试</text><line x1="504" y1="138" x2="516" y2="138" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-evaluation)"/><rect x="516" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="591.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="591.0" y="138.0">迁移</tspan></text><text x="591.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">迁移测试(无 AI)</text></svg><div class="visual-suppressed"><strong>基准图已抑制</strong><p>result.json 未携带 benchmark.baselines,本图不绘制;基准表现见独立基准报告。</p></div></div></section><section id="full-06-sources" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>06 来源、溯源与附录</h2><p class="full-chapter-lead">保留原始来源、URL、证据 ID 和获取信息,确保可回查。</p></header><div class="full-chapter-body"><h3>来源列表</h3><div class='table-wrap'><table class='data-table source-table'><thead><tr><th>ID</th><th>标题</th><th>年份</th><th>权威级别</th><th>可验证位置</th></tr></thead><tbody><tr><td><code>S-2023-kazemitabaar</code></td><td class='cell-main source-title-cell'>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>查看来源与溯源</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>原文标题</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="source-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="source-detail-row"><dt>权威级别</dt><dd>T1 DOI 可验证论文</dd></div><div class="source-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3544548.3580919</dd></div><div class="source-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">https://dl.acm.org/doi/10.1145/3544548.3580919</a></dd></div></dl></details></td><td>2023</td><td>T1 DOI 可验证论文</td><td class='cell-main'><a href='https://dl.acm.org/doi/10.1145/3544548.3580919'>https://dl.acm.org/doi/10.1145/3544548.3580919</a></td></tr><tr><td><code>S-2025-bastani</code></td><td class='cell-main source-title-cell'>Generative AI without guardrails can harm learning: Evidence from high school mathematics <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>查看来源与溯源</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>原文标题</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="source-detail-row"><dt>年份</dt><dd>2025</dd></div><div class="source-detail-row"><dt>权威级别</dt><dd>T1 DOI 可验证论文</dd></div><div class="source-detail-row"><dt>来源位置</dt><dd>https://www.pnas.org/doi/10.1073/pnas.2422633122</dd></div><div class="source-detail-row"><dt>可验证链接</dt><dd><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">https://www.pnas.org/doi/10.1073/pnas.2422633122</a></dd></div></dl></details></td><td>2025</td><td>T1 DOI 可验证论文</td><td class='cell-main'><a href='https://www.pnas.org/doi/10.1073/pnas.2422633122'>https://www.pnas.org/doi/10.1073/pnas.2422633122</a></td></tr><tr><td><code>S-2024-marzuki</code></td><td class='cell-main source-title-cell'>Impact of ChatGPT on ESL students&#x27; academic writing skills <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>查看来源与溯源</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>原文标题</dt><dd>Impact of ChatGPT on ESL students&#x27; academic writing skills</dd></div><div class="source-detail-row"><dt>年份</dt><dd>2024</dd></div><div class="source-detail-row"><dt>权威级别</dt><dd>T1 DOI 可验证论文</dd></div><div class="source-detail-row"><dt>来源位置</dt><dd>https://link.springer.com/article/10.1186/s40561-024-00295-9</dd></div><div class="source-detail-row"><dt>可验证链接</dt><dd><a href="https://link.springer.com/article/10.1186/s40561-024-00295-9">https://link.springer.com/article/10.1186/s40561-024-00295-9</a></dd></div></dl></details></td><td>2024</td><td>T1 DOI 可验证论文</td><td class='cell-main'><a href='https://link.springer.com/article/10.1186/s40561-024-00295-9'>https://link.springer.com/article/10.1186/s40561-024-00295-9</a></td></tr><tr><td><code>S-2023-peng</code></td><td class='cell-main source-title-cell'>The Impact of AI on Developer Productivity: Evidence from GitHub Copilot <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>查看来源与溯源</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>原文标题</dt><dd>The Impact of AI on Developer Productivity: Evidence from GitHub Copilot</dd></div><div class="source-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="source-detail-row"><dt>权威级别</dt><dd>tier2_academic_database</dd></div><div class="source-detail-row"><dt>来源位置</dt><dd>https://arxiv.org/abs/2302.06590</dd></div><div class="source-detail-row"><dt>可验证链接</dt><dd><a href="https://doi.org/10.48550/arXiv.2302.06590">https://doi.org/10.48550/arXiv.2302.06590</a></dd></div></dl></details></td><td>2023</td><td>tier2_academic_database</td><td class='cell-main'><a href='https://doi.org/10.48550/arXiv.2302.06590'>https://doi.org/10.48550/arXiv.2302.06590</a></td></tr><tr><td><code>S-2023-yetistiren</code></td><td class='cell-main source-title-cell'>GitHub Copilot AI Pair Programmer: Asset or Liability? <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>查看来源与溯源</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>原文标题</dt><dd>GitHub Copilot AI Pair Programmer: Asset or Liability?</dd></div><div class="source-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="source-detail-row"><dt>权威级别</dt><dd>T1 DOI 可验证论文</dd></div><div class="source-detail-row"><dt>来源位置</dt><dd>https://doi.org/10.1016/j.jss.2023.111734</dd></div><div class="source-detail-row"><dt>可验证链接</dt><dd><a href="https://doi.org/10.1016/j.jss.2023.111734">https://doi.org/10.1016/j.jss.2023.111734</a></dd></div></dl></details></td><td>2023</td><td>T1 DOI 可验证论文</td><td class='cell-main'><a href='https://doi.org/10.1016/j.jss.2023.111734'>https://doi.org/10.1016/j.jss.2023.111734</a></td></tr><tr><td><code>S-2022-finnie-ansley</code></td><td class='cell-main source-title-cell'>Using GitHub Copilot to Solve Introductory Programming Problems <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>查看来源与溯源</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>原文标题</dt><dd>Using GitHub Copilot to Solve Introductory Programming Problems</dd></div><div class="source-detail-row"><dt>年份</dt><dd>2022</dd></div><div class="source-detail-row"><dt>权威级别</dt><dd>T1 DOI 可验证论文</dd></div><div class="source-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3545945.3569830</dd></div><div class="source-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3545945.3569830">https://dl.acm.org/doi/10.1145/3545945.3569830</a></dd></div></dl></details></td><td>2022</td><td>T1 DOI 可验证论文</td><td class='cell-main'><a href='https://dl.acm.org/doi/10.1145/3545945.3569830'>https://dl.acm.org/doi/10.1145/3545945.3569830</a></td></tr><tr><td><code>S-2023-explanations-compare</code></td><td class='cell-main source-title-cell'>Comparing Code Explanations Created by Students and Large Language Models <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>查看来源与溯源</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>原文标题</dt><dd>Comparing Code Explanations Created by Students and Large Language Models</dd></div><div class="source-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="source-detail-row"><dt>权威级别</dt><dd>T1 DOI 可验证论文</dd></div><div class="source-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3587102.3588785</dd></div><div class="source-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3587102.3588785">https://dl.acm.org/doi/10.1145/3587102.3588785</a></dd></div></dl></details></td><td>2023</td><td>T1 DOI 可验证论文</td><td class='cell-main'><a href='https://dl.acm.org/doi/10.1145/3587102.3588785'>https://dl.acm.org/doi/10.1145/3587102.3588785</a></td></tr><tr><td><code>S-2022-vaithilingam</code></td><td class='cell-main source-title-cell'>Expectation vs. Experience: Evaluating the Usability of Code Generation Tools <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>查看来源与溯源</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>原文标题</dt><dd>Expectation vs. Experience: Evaluating the Usability of Code Generation Tools</dd></div><div class="source-detail-row"><dt>年份</dt><dd>2022</dd></div><div class="source-detail-row"><dt>权威级别</dt><dd>T1 DOI 可验证论文</dd></div><div class="source-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3491101.3519665</dd></div><div class="source-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3491101.3519665">https://dl.acm.org/doi/10.1145/3491101.3519665</a></dd></div></dl></details></td><td>2022</td><td>T1 DOI 可验证论文</td><td class='cell-main'><a href='https://dl.acm.org/doi/10.1145/3491101.3519665'>https://dl.acm.org/doi/10.1145/3491101.3519665</a></td></tr></tbody></table></div><h3>Fetch 溯源</h3><p class="provenance-summary">搜索提供方:n/a</p><p class='provenance-empty'>无逐条 fetch 记录(来源由研究管线直接提供)。</p></div></section></main></div>
2222
+ <svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="评价设计流程"><title>评价设计流程</title><desc>基线、后测、保持测试与迁移测试的评价流程;完整指标与分析计划见评估章节。</desc><defs><marker id="arr-zh-full-evaluation" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">评价设计流程</text><rect x="30" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="105.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="105.0" y="138.0">基线</tspan></text><text x="105.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">前测</text><line x1="180" y1="138" x2="192" y2="138" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-evaluation)"/><rect x="192" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="267.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="267.0" y="138.0">后测</tspan></text><text x="267.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">后测</text><line x1="342" y1="138" x2="354" y2="138" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-evaluation)"/><rect x="354" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="429.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="429.0" y="138.0">保持</tspan></text><text x="429.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">保持测试</text><line x1="504" y1="138" x2="516" y2="138" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-evaluation)"/><rect x="516" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="591.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="591.0" y="138.0">迁移</tspan></text><text x="591.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">迁移测试(无 AI)</text></svg><div class="visual-suppressed"><strong>基准图已抑制</strong><p>result.json 未携带 benchmark.baselines,本图不绘制;基准表现见独立基准报告。</p></div></div></section><section id="full-06-sources" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>06 来源、溯源与附录</h2><p class="full-chapter-lead">保留原始来源、URL、证据 ID 和获取信息,确保可回查。</p></header><div class="full-chapter-body"><h3>来源列表</h3><div class='table-wrap'><table class='data-table source-table'><thead><tr><th>ID</th><th>标题</th><th>年份</th><th>权威级别</th><th>可验证位置</th></tr></thead><tbody><tr><td><code>S-2023-kazemitabaar</code></td><td class='cell-main source-title-cell'>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>查看来源与溯源</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>原文标题</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="source-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="source-detail-row"><dt>权威级别</dt><dd>T1 DOI 可验证论文</dd></div><div class="source-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3544548.3580919</dd></div><div class="source-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">https://dl.acm.org/doi/10.1145/3544548.3580919</a></dd></div></dl></details></td><td>2023</td><td>T1 DOI 可验证论文</td><td class='cell-main'><a href='https://dl.acm.org/doi/10.1145/3544548.3580919'>https://dl.acm.org/doi/10.1145/3544548.3580919</a></td></tr><tr><td><code>S-2025-bastani</code></td><td class='cell-main source-title-cell'>Generative AI without guardrails can harm learning: Evidence from high school mathematics <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>查看来源与溯源</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>原文标题</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="source-detail-row"><dt>年份</dt><dd>2025</dd></div><div class="source-detail-row"><dt>权威级别</dt><dd>T1 DOI 可验证论文</dd></div><div class="source-detail-row"><dt>来源位置</dt><dd>https://www.pnas.org/doi/10.1073/pnas.2422633122</dd></div><div class="source-detail-row"><dt>可验证链接</dt><dd><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">https://www.pnas.org/doi/10.1073/pnas.2422633122</a></dd></div></dl></details></td><td>2025</td><td>T1 DOI 可验证论文</td><td class='cell-main'><a href='https://www.pnas.org/doi/10.1073/pnas.2422633122'>https://www.pnas.org/doi/10.1073/pnas.2422633122</a></td></tr><tr><td><code>S-2024-marzuki</code></td><td class='cell-main source-title-cell'>Impact of ChatGPT on ESL students&#x27; academic writing skills <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>查看来源与溯源</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>原文标题</dt><dd>Impact of ChatGPT on ESL students&#x27; academic writing skills</dd></div><div class="source-detail-row"><dt>年份</dt><dd>2024</dd></div><div class="source-detail-row"><dt>权威级别</dt><dd>T1 DOI 可验证论文</dd></div><div class="source-detail-row"><dt>来源位置</dt><dd>https://link.springer.com/article/10.1186/s40561-024-00295-9</dd></div><div class="source-detail-row"><dt>可验证链接</dt><dd><a href="https://link.springer.com/article/10.1186/s40561-024-00295-9">https://link.springer.com/article/10.1186/s40561-024-00295-9</a></dd></div></dl></details></td><td>2024</td><td>T1 DOI 可验证论文</td><td class='cell-main'><a href='https://link.springer.com/article/10.1186/s40561-024-00295-9'>https://link.springer.com/article/10.1186/s40561-024-00295-9</a></td></tr><tr><td><code>S-2023-peng</code></td><td class='cell-main source-title-cell'>The Impact of AI on Developer Productivity: Evidence from GitHub Copilot <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>查看来源与溯源</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>原文标题</dt><dd>The Impact of AI on Developer Productivity: Evidence from GitHub Copilot</dd></div><div class="source-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="source-detail-row"><dt>权威级别</dt><dd>Tier2 Academic Database</dd></div><div class="source-detail-row"><dt>来源位置</dt><dd>https://arxiv.org/abs/2302.06590</dd></div><div class="source-detail-row"><dt>可验证链接</dt><dd><a href="https://doi.org/10.48550/arXiv.2302.06590">https://doi.org/10.48550/arXiv.2302.06590</a></dd></div></dl></details></td><td>2023</td><td>Tier2 Academic Database</td><td class='cell-main'><a href='https://doi.org/10.48550/arXiv.2302.06590'>https://doi.org/10.48550/arXiv.2302.06590</a></td></tr><tr><td><code>S-2023-yetistiren</code></td><td class='cell-main source-title-cell'>GitHub Copilot AI Pair Programmer: Asset or Liability? <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>查看来源与溯源</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>原文标题</dt><dd>GitHub Copilot AI Pair Programmer: Asset or Liability?</dd></div><div class="source-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="source-detail-row"><dt>权威级别</dt><dd>T1 DOI 可验证论文</dd></div><div class="source-detail-row"><dt>来源位置</dt><dd>https://doi.org/10.1016/j.jss.2023.111734</dd></div><div class="source-detail-row"><dt>可验证链接</dt><dd><a href="https://doi.org/10.1016/j.jss.2023.111734">https://doi.org/10.1016/j.jss.2023.111734</a></dd></div></dl></details></td><td>2023</td><td>T1 DOI 可验证论文</td><td class='cell-main'><a href='https://doi.org/10.1016/j.jss.2023.111734'>https://doi.org/10.1016/j.jss.2023.111734</a></td></tr><tr><td><code>S-2022-finnie-ansley</code></td><td class='cell-main source-title-cell'>Using GitHub Copilot to Solve Introductory Programming Problems <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>查看来源与溯源</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>原文标题</dt><dd>Using GitHub Copilot to Solve Introductory Programming Problems</dd></div><div class="source-detail-row"><dt>年份</dt><dd>2022</dd></div><div class="source-detail-row"><dt>权威级别</dt><dd>T1 DOI 可验证论文</dd></div><div class="source-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3545945.3569830</dd></div><div class="source-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3545945.3569830">https://dl.acm.org/doi/10.1145/3545945.3569830</a></dd></div></dl></details></td><td>2022</td><td>T1 DOI 可验证论文</td><td class='cell-main'><a href='https://dl.acm.org/doi/10.1145/3545945.3569830'>https://dl.acm.org/doi/10.1145/3545945.3569830</a></td></tr><tr><td><code>S-2023-explanations-compare</code></td><td class='cell-main source-title-cell'>Comparing Code Explanations Created by Students and Large Language Models <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>查看来源与溯源</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>原文标题</dt><dd>Comparing Code Explanations Created by Students and Large Language Models</dd></div><div class="source-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="source-detail-row"><dt>权威级别</dt><dd>T1 DOI 可验证论文</dd></div><div class="source-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3587102.3588785</dd></div><div class="source-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3587102.3588785">https://dl.acm.org/doi/10.1145/3587102.3588785</a></dd></div></dl></details></td><td>2023</td><td>T1 DOI 可验证论文</td><td class='cell-main'><a href='https://dl.acm.org/doi/10.1145/3587102.3588785'>https://dl.acm.org/doi/10.1145/3587102.3588785</a></td></tr><tr><td><code>S-2022-vaithilingam</code></td><td class='cell-main source-title-cell'>Expectation vs. Experience: Evaluating the Usability of Code Generation Tools <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>查看来源与溯源</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>原文标题</dt><dd>Expectation vs. Experience: Evaluating the Usability of Code Generation Tools</dd></div><div class="source-detail-row"><dt>年份</dt><dd>2022</dd></div><div class="source-detail-row"><dt>权威级别</dt><dd>T1 DOI 可验证论文</dd></div><div class="source-detail-row"><dt>来源位置</dt><dd>https://dl.acm.org/doi/10.1145/3491101.3519665</dd></div><div class="source-detail-row"><dt>可验证链接</dt><dd><a href="https://dl.acm.org/doi/10.1145/3491101.3519665">https://dl.acm.org/doi/10.1145/3491101.3519665</a></dd></div></dl></details></td><td>2022</td><td>T1 DOI 可验证论文</td><td class='cell-main'><a href='https://dl.acm.org/doi/10.1145/3491101.3519665'>https://dl.acm.org/doi/10.1145/3491101.3519665</a></td></tr></tbody></table></div><h3>Fetch 溯源</h3><p class="provenance-summary">搜索提供方:n/a</p><p class='provenance-empty'>无逐条 fetch 记录(来源由研究管线直接提供)。</p></div></section></main></div>
2063
2223
  </div>
2064
2224
  <footer class="report-footer"><p>EduEvidence 证据报告 · Schema PASS · Claim Binding PASS · Numeric Consistency PASS · Bilingual Structure PASS · 语言人话化 PASS · 无伪精度 PASS · Lieflat 数据溯源 PASS · 坐标轴无失真 NOT_CHECKED · 色盲安全 NOT_CHECKED · 单文件离线可打开 · 数据源:result.json</p></footer>
2065
2225
  </div>
@@ -2082,64 +2242,141 @@ select:focus-visible,
2082
2242
  </div>
2083
2243
  <p class="hero-rationale">Positive task-performance evidence plus documented unguarded-access risk, mixed quality/usability signals, and missing university-level learning evidence → bounded, guardrailed pilot with evaluation, not full adoption.</p>
2084
2244
  <div class="hero-insights">
2085
- <article class="hero-insight support"><span>Strongest supported conclusion</span><p class="hero-insight-text">AI coding assistants raise task performance for novices during training.</p></article>
2086
- <article class="hero-insight uncertain"><span>Key uncertainty / contradiction</span><p class="hero-insight-text">Whether AI coding assistants improve or preserve actual programming learning in university novices — no direct university-level RCT in reviewed set [无直接证据]</p></article>
2087
- <article class="hero-insight risk"><span>Main risk</span><p class="hero-insight-text">AI dependency and over-reliance risk is real and documented for unguarded usage (E-004) and foreshadowed by usability findings (E-012).</p></article>
2088
- <article class="hero-insight next"><span>Next action</span><p class="hero-insight-text">Not provided in this research result.</p></article>
2245
+ <article class="hero-insight support"><span>Strongest supported conclusion</span><p class="hero-insight-text">AI coding assistants reliably speed up practice work: completion rate 1.15x and time 0.57x in a randomised trial of 69 novices.</p></article>
2246
+ <article class="hero-insight uncertain"><span>Key uncertainty / contradiction</span><p class="hero-insight-text">No university-level RCT measures learning directly, and the one large trial that did - unguarded GPT-4 - saw independent exam scores fall 17%.</p></article>
2247
+ <article class="hero-insight risk"><span>Main risk</span><p class="hero-insight-text">Unguarded access can raise practice performance while lowering independent exam performance, and learners may not notice the gap.</p></article>
2248
+ <article class="hero-insight next"><span>Next action</span><p class="hero-insight-text">Run a phased CS1 pilot with hints-not-answers guardrails, weekly lab use, and a no-AI transfer exam that can stop the pilot.</p></article>
2089
2249
  </div>
2090
2250
  <p class="hero-provenance"><span>Evidence / sources</span> · 12 / 8</p>
2091
- </div></div></section><section class="brief-block brief-outcomes" id="brief-outcomes-en"><header class="brief-block-header"><h2>Task performance ≠ learning</h2><p>Only informative outcome separation; positive, negative and null effects use effect_direction.</p></header><div class="brief-block-body"><div class="outcome-separation" data-visual="outcome-separation"><div class="visual-heading"><h3>Outcome Separation · Task performance ≠ learning</h3><p>Outcomes are adjudicated separately so faster training performance is not silently treated as evidence of learning.</p><p class="semantic-note">Effect direction comes from evidence.effect_direction; supporting a claim does not imply a positive outcome.</p></div><div class="outcome-groups"><article class="outcome-group outcome-task"><h3>Task / proximal performance</h3><ul><li><strong>Completion time</strong><span class="outcome-states"><span class="dir pos">Positive effect 2</span></span></li><li><strong>Code quality</strong><span class="outcome-states"><span class="dir neu">Null effect 1</span></span></li><li><strong>Assignment score</strong><span class="outcome-states"><span class="dir pos">Positive effect 2</span></span></li></ul></article><article class="outcome-group outcome-learning"><h3>Learning / retention / transfer</h3><ul><li><strong>Knowledge gain</strong><span class="outcome-states"><span class="dir pos">Positive effect 1</span></span></li><li><strong>Retention</strong><span class="outcome-states"><span class="dir neu">Null effect 1</span></span></li><li><strong>Independent problem solving</strong><span class="outcome-states"><span class="dir neg">Negative effect 1</span><span class="dir neu">Null effect 2</span></span></li></ul></article><article class="outcome-group outcome-risk"><h3>Risk / dependency</h3><ul><li><strong>Over-reliance</strong><span class="outcome-states"><span class="dir neg">Negative effect 1</span></span></li></ul></article><article class="outcome-group outcome-other"><h3>Other outcomes</h3><ul><li><strong>Metacognition</strong><span class="outcome-states"><span class="dir pos">Positive effect 1</span></span></li></ul></article></div></div><div class="visual-surface brief-chart" data-visual="outcome-evidence-balance"><svg viewBox="0 0 720 300" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="Outcome evidence effect balance"><title>Outcome evidence effect balance</title><desc>Positive / negative / null effect counts per outcome (based on effect_direction, not claim support).</desc><rect x="0" y="0" width="720" height="300" fill="#FFFFFF"/><line x1="415.0" y1="46" x2="415.0" y2="182" stroke="#999" stroke-width="1" stroke-dasharray="3,3"/><text x="142" y="58.0" text-anchor="end" font-size="11" fill="#333">Knowledge gain</text><rect x="415.0" y="49.8" width="132.5" height="9.4" fill="#5E8A6A"/><text x="481.2" y="58.0" text-anchor="middle" font-size="9" fill="#fff">1</text><text x="142" y="75.0" text-anchor="end" font-size="11" fill="#333">Retention</text><text x="142" y="92.0" text-anchor="end" font-size="11" fill="#333">Independent problem solving</text><rect x="282.5" y="83.8" width="132.5" height="9.4" fill="#A85B53"/><text x="348.8" y="92.0" text-anchor="middle" font-size="9" fill="#fff">-1</text><text x="142" y="109.0" text-anchor="end" font-size="11" fill="#333">Completion time</text><rect x="415.0" y="100.8" width="265.0" height="9.4" fill="#5E8A6A"/><text x="547.5" y="109.0" text-anchor="middle" font-size="9" fill="#fff">2</text><text x="142" y="126.0" text-anchor="end" font-size="11" fill="#333">Code quality</text><text x="142" y="143.0" text-anchor="end" font-size="11" fill="#333">Assignment score</text><rect x="415.0" y="134.8" width="265.0" height="9.4" fill="#5E8A6A"/><text x="547.5" y="143.0" text-anchor="middle" font-size="9" fill="#fff">2</text><text x="142" y="160.0" text-anchor="end" font-size="11" fill="#333">Metacognition</text><rect x="415.0" y="151.8" width="132.5" height="9.4" fill="#5E8A6A"/><text x="481.2" y="160.0" text-anchor="middle" font-size="9" fill="#fff">1</text><text x="142" y="177.0" text-anchor="end" font-size="11" fill="#333">Over-reliance</text><rect x="282.5" y="168.8" width="132.5" height="9.4" fill="#A85B53"/><text x="348.8" y="177.0" text-anchor="middle" font-size="9" fill="#fff">-1</text><rect x="415.0" y="206.1" width="8.0" height="6" fill="#C99A4A"/><rect x="415.0" y="214.9" width="8.0" height="6" fill="#C99A4A"/><rect x="415.0" y="232.4" width="8.0" height="6" fill="#C99A4A"/><rect x="150" y="14" width="10" height="10" fill="#5E8A6A"/><text x="164" y="23" font-size="10" fill="#333">Positive effect</text><rect x="347" y="14" width="10" height="10" fill="#A85B53"/><text x="361" y="23" font-size="10" fill="#333">Negative effect</text><rect x="544" y="14" width="10" height="10" fill="#C99A4A"/><text x="558" y="23" font-size="10" fill="#333">Null effect</text></svg><p class="chart-interpretation"><strong>What this means: </strong>Positive / negative / null effect-direction evidence counts per outcome. This visual encodes effect_direction, not whether evidence supports a claim.</p></div></div></section><section class="brief-block brief-tribunal" id="brief-tribunal-en"><header class="brief-block-header"><h2>Evidence tribunal</h2><p>Supported, uncertain, contradicted and missing evidence stay separated instead of flattened into long prose.</p></header><div class="brief-block-body"><div class="evidence-tribunal" data-visual="evidence-tribunal-grid"><p class="tribunal-summary"><strong>Decision: </strong>Pilot · <strong>Confidence: </strong>Moderate</p><div class="tribunal-grid"><article class="tribunal-card supported"><header><span aria-hidden="true">✓</span><h3>Can claim</h3><span class="tribunal-count">7</span></header><ul><li><p>AI coding assistants reliably increase task performance during training (completion speed, correctness) — E-001, E-006.</p><div class="tribunal-evidence-refs"><code>E-001</code><code>E-006</code></div></li><li><p>Unguarded generative AI access can harm independent problem solving when access is removed — E-004.</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li><li><p>Guardrail design (hints instead of answers) substantially mitigates the negative learning effect — E-005.</p><div class="tribunal-evidence-refs"><code>E-005</code></div></li></ul><details class="tribunal-more"><summary>View 4 more</summary><ul><li><p>Task performance gains do not automatically imply learning gains — E-004 vs E-006 (within-study contrast).</p><div class="tribunal-evidence-refs"><code>E-004</code><code>E-006</code></div></li><li><p>Tool capability is substantial: Codex solves roughly half to three-quarters of CS1 exam-style questions — E-010.</p><div class="tribunal-evidence-refs"><code>E-010</code></div></li><li><p>Professional-developer RCT shows ~55% faster task completion with Copilot; directness limited by professional population — E-008.</p><div class="tribunal-evidence-refs"><code>E-008</code></div></li><li><p>LLM code explanations rate comparable to student-authored explanations, viable as scaffold material — E-011.</p><div class="tribunal-evidence-refs"><code>E-011</code></div></li></ul></details></article><article class="tribunal-card uncertain"><header><span aria-hidden="true">?</span><h3>Cannot yet claim</h3><span class="tribunal-count">4</span></header><ul><li><p>Whether AI coding assistants improve or preserve actual programming learning in university novices — no direct university-level RCT in reviewed set [无直接证据]</p></li><li><p>Whether one-week neutral retention (Kazemitabaar 2023) extends to a semester — E-003.</p><div class="tribunal-evidence-refs"><code>E-003</code></div></li><li><p>Whether benchmark quality findings (E-009) and explanation-quality ratings (E-011) translate into classroom learning gains.</p><div class="tribunal-evidence-refs"><code>E-009</code><code>E-011</code></div></li></ul><details class="tribunal-more"><summary>View 1 more</summary><ul><li><p>How comprehension/ownership difficulties documented in usability studies (E-012) behave over a full semester with guardrails.</p><div class="tribunal-evidence-refs"><code>E-012</code></div></li></ul></details></article><article class="tribunal-card contradicted"><header><span aria-hidden="true">×</span><h3>Contradicted claims</h3><span class="tribunal-count">2</span></header><ul><li><p>The claim &#x27;AI tools always improve learning&#x27; is contradicted by E-004 (unguarded access, -17% independent exam).</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li><li><p>The claim &#x27;speed gains equal learning gains&#x27; is contradicted by the task-vs-learning separation across E-001/E-006/E-008 vs E-004.</p><div class="tribunal-evidence-refs"><code>E-001</code><code>E-006</code><code>E-008</code><code>E-004</code></div></li></ul></article><article class="tribunal-card missing"><header><span aria-hidden="true">…</span><h3>Missing evidence</h3><span class="tribunal-count">4</span></header><ul><li><p>RCT of AI coding assistants in university programming courses with retention and no-AI transfer tests.</p></li><li><p>Studies varying AI usage policy within the same course.</p></li><li><p>Longitudinal data on AI dependency beyond one course.</p></li></ul><details class="tribunal-more"><summary>View 1 more</summary><ul><li><p>Peer-reviewed replication of the professional speed RCT (Peng et al. remains a preprint).</p></li></ul></details></article></div></div></div></section><section class="brief-block brief-action" id="brief-action-en"><header class="brief-block-header"><h2>Evidence to action</h2><p>Applicability, guardrails, stop conditions and evaluation form one executable path.</p></header><div class="brief-block-body"><div class="evidence-to-action" data-visual="evidence-to-action"><article class="action-node action-evidence"><span>Evidence</span><p class="action-node-text">AI coding assistants reliably increase task performance during training (completion speed, correctness) — E-001, E-006.</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-applicability"><span>Applicability</span><p class="action-node-text">pilot in first-year C course with guardrailed usage policy</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-decision"><span>Decision</span><p class="action-node-text">Pilot</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-guardrails"><span>Guardrails</span><div class="action-node-text"><p>AI usage is allowed in three explicitly graded modes (explain / collaborate / no-AI-transfer).…</p><details class="detail-expander"><summary>Expand full explanation</summary><div class="detail-body"><p>AI usage is allowed in three explicitly graded modes (explain / collaborate / no-AI-transfer). Copying unexamined AI output is an academic integrity violation and is assessed via the reasoning-trace requirement.</p></div></details></div></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-stop"><span>Stop conditions</span><div class="action-node-text"><p>transfer-test scores drop significantly below baseline cohort expectations; widespread integrity violations in reasoning traces;…</p><details class="detail-expander"><summary>Expand full explanation</summary><div class="detail-body"><p>transfer-test scores drop significantly below baseline cohort expectations; widespread integrity violations in reasoning traces; AI dependency signals exceed threshold in risk metrics; TA/teacher workload becomes unsustainable</p></div></details></div></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-evaluation"><span>Evaluation</span><div class="action-node-text"><p>treatment group shows non-inferior independent problem solving (delta within 5%) AND superior or equal retention AND AI dependency index below thresho…</p><details class="detail-expander"><summary>Expand full explanation</summary><div class="detail-body"><p>treatment group shows non-inferior independent problem solving (delta within 5%) AND superior or equal retention AND AI dependency index below threshold; if independent problem solving declines &gt;10%, the pilot is judged unsuccessful regardless of task-performance gains.</p></div></details></div></article></div></div></section><section class="brief-block brief-lieflat" id="brief-lieflat-en"><header class="brief-block-header"><h2>Lieflat Editorial Gallery</h2><p>Charts selected and composed by AI from the Lieflat catalog; every number traces back to result.json.</p></header><div class="brief-block-body"><div class="lieflat-gallery-container"><figure class="lieflat-card" data-lieflat data-visual="lieflat-bubble_almanac" data-chart-id="lieflat-bubble-almanac.svg"><h3 class="lieflat-title">Year × dimension evidence almanac</h3><p class="lieflat-sub">Bubble area ∝ study count (sqrt) · solid core = significant results</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 540 300" width="100%" height="100%" role="img" aria-label="Year × dimension evidence almanac" style="background:#111827;">
2092
- <line x1="44" y1="70" x2="520" y2="70" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:0ms"/>
2093
- <line x1="44" y1="77" x2="520" y2="77" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:14ms"/>
2094
- <line x1="44" y1="84" x2="520" y2="84" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:28ms"/>
2095
- <line x1="44" y1="91" x2="520" y2="91" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:42ms"/>
2096
- <line x1="44" y1="98" x2="520" y2="98" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:56ms"/>
2097
- <line x1="44" y1="105" x2="520" y2="105" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:70ms"/>
2098
- <line x1="44" y1="112" x2="520" y2="112" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:84ms"/>
2099
- <line x1="44" y1="119" x2="520" y2="119" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:98ms"/>
2100
- <line x1="44" y1="126" x2="520" y2="126" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:112ms"/>
2101
- <line x1="44" y1="133" x2="520" y2="133" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:126ms"/>
2102
- <line x1="44" y1="140" x2="520" y2="140" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:140ms"/>
2103
- <line x1="44" y1="147" x2="520" y2="147" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:154ms"/>
2104
- <line x1="44" y1="154" x2="520" y2="154" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:168ms"/>
2105
- <line x1="44" y1="161" x2="520" y2="161" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:182ms"/>
2106
- <line x1="44" y1="168" x2="520" y2="168" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:196ms"/>
2107
- <line x1="44" y1="175" x2="520" y2="175" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:210ms"/>
2108
- <line x1="44" y1="182" x2="520" y2="182" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:224ms"/>
2109
- <line x1="44" y1="189" x2="520" y2="189" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:238ms"/>
2110
- <line x1="44" y1="196" x2="520" y2="196" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:252ms"/>
2111
- <line x1="44" y1="203" x2="520" y2="203" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:266ms"/>
2112
- <line x1="44" y1="210" x2="520" y2="210" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:280ms"/>
2113
- <line x1="44" y1="217" x2="520" y2="217" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:294ms"/>
2114
- <line x1="44" y1="224" x2="520" y2="224" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:308ms"/>
2115
- <line x1="44" y1="231" x2="520" y2="231" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:322ms"/>
2116
- <line x1="44" y1="238" x2="520" y2="238" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:336ms"/>
2117
- <line x1="44" y1="245" x2="520" y2="245" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:350ms"/>
2118
- <line x1="44" y1="252" x2="520" y2="252" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:364ms"/>
2119
- <line x1="44" y1="259" x2="520" y2="259" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:378ms"/>
2120
- <text x="150" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:40ms">completion tim</text>
2121
- <text x="203" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:140ms">independent pr</text>
2122
- <text x="256" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:240ms">retention</text>
2123
- <text x="309" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:340ms">assignment sco</text>
2124
- <text x="361" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:440ms">knowledge gain</text>
2125
- <text x="414" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:540ms">code quality</text>
2126
- <text x="467" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:640ms">metacognition</text>
2127
- <text x="520" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:740ms">over reliance</text>
2128
- <text x="96" y="96" fill="#94A3B8" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:80ms">2022</text>
2129
- <text x="96" y="140" fill="#94A3B8" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:180ms">2023</text>
2130
- <text x="96" y="184" fill="#94A3B8" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:280ms">2024</text>
2131
- <text x="96" y="228" fill="#94A3B8" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:380ms">2025</text>
2132
- <circle cx="308.6" cy="92.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:100ms"><title>assignment_score (2022) — N = 1 studies, significant = 0</title></circle>
2133
- <circle cx="520.0" cy="92.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:112ms"><title>over_reliance (2022) — N = 1 studies, significant = 0</title></circle>
2134
- <circle cx="150.0" cy="136.0" r="5.1" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:124ms"><title>completion_time (2023) — N = 2 studies, significant = 0</title></circle>
2135
- <circle cx="202.9" cy="136.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:136ms"><title>independent_problem_solving (2023) — N = 1 studies, significant = 0</title></circle>
2136
- <circle cx="255.7" cy="136.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:148ms"><title>retention (2023) — N = 1 studies, significant = 0</title></circle>
2137
- <circle cx="414.3" cy="136.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:160ms"><title>code_quality (2023) — N = 1 studies, significant = 0</title></circle>
2138
- <circle cx="467.1" cy="136.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:172ms"><title>metacognition (2023) — N = 1 studies, significant = 0</title></circle>
2139
- <circle cx="361.4" cy="180.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:184ms"><title>knowledge_gain (2024) — N = 1 studies, significant = 0</title></circle>
2140
- <circle cx="202.9" cy="224.0" r="5.1" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:196ms"><title>independent_problem_solving (2025) — N = 2 studies, significant = 0</title></circle>
2141
- <circle cx="308.6" cy="224.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:208ms"><title>assignment_score (2025) — N = 1 studies, significant = 0</title></circle>
2142
- </svg></div><figcaption class="lieflat-caption">Drawn only when years and outcome dimensions exist.</figcaption><p class="lieflat-src">L9 Bubble Almanac · evidence.year_x_dimension</p></figure><figure class="lieflat-card" data-lieflat data-visual="lieflat-tick_rows" data-chart-id="lieflat-tick-rows.svg"><h3 class="lieflat-title">Effect direction by outcome</h3><p class="lieflat-sub">One dot = one evidence item · green = positive · grey = null · orange = negative · right number = net</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 540 300" width="100%" height="100%" role="img" aria-label="Effect direction by outcome" style="background:#111827;">
2251
+ </div></div></section><section class="brief-block brief-outcomes" id="brief-outcomes-en"><header class="brief-block-header"><h2>Task performance ≠ learning</h2><p>Only informative outcome separation; positive, negative and null effects use effect_direction.</p></header><div class="brief-block-body"><div class="outcome-separation" data-visual="outcome-separation"><div class="visual-heading"><h3>Outcome Separation · Task performance ≠ learning</h3><p>Outcomes are adjudicated separately so faster training performance is not silently treated as evidence of learning.</p><p class="semantic-note">Effect direction comes from evidence.effect_direction; supporting a claim does not imply a positive outcome.</p></div><div class="outcome-groups"><article class="outcome-group outcome-task"><h3>Task / proximal performance</h3><ul><li><strong>Completion time</strong><span class="outcome-states"><span class="dir pos">Positive effect 2</span></span></li><li><strong>Code quality</strong><span class="outcome-states"><span class="dir neu">Null effect 1</span></span></li><li><strong>Assignment score</strong><span class="outcome-states"><span class="dir pos">Positive effect 2</span></span></li></ul></article><article class="outcome-group outcome-learning"><h3>Learning / retention / transfer</h3><ul><li><strong>Knowledge gain</strong><span class="outcome-states"><span class="dir pos">Positive effect 1</span></span></li><li><strong>Retention</strong><span class="outcome-states"><span class="dir neu">Null effect 1</span></span></li><li><strong>Independent problem solving</strong><span class="outcome-states"><span class="dir neg">Negative effect 1</span><span class="dir neu">Null effect 2</span></span></li></ul></article><article class="outcome-group outcome-risk"><h3>Risk / dependency</h3><ul><li><strong>Over-reliance</strong><span class="outcome-states"><span class="dir neg">Negative effect 1</span></span></li></ul></article><article class="outcome-group outcome-other"><h3>Other outcomes</h3><ul><li><strong>Metacognition</strong><span class="outcome-states"><span class="dir pos">Positive effect 1</span></span></li></ul></article></div></div><div class="visual-surface brief-chart" data-visual="outcome-evidence-balance"><svg viewBox="0 0 720 300" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="Outcome evidence effect balance"><title>Outcome evidence effect balance</title><desc>Positive / negative / null effect counts per outcome (based on effect_direction, not claim support).</desc><rect x="0" y="0" width="720" height="300" fill="#FFFFFF"/><line x1="439.0" y1="46" x2="439.0" y2="182" stroke="#999" stroke-width="1" stroke-dasharray="3,3"/><text x="190" y="58.0" text-anchor="end" font-size="11" fill="#333">Knowledge gain</text><rect x="439.0" y="49.8" width="120.5" height="9.4" fill="#5E8A6A"/><text x="499.2" y="58.0" text-anchor="middle" font-size="9" fill="#fff">1</text><text x="190" y="75.0" text-anchor="end" font-size="11" fill="#333">Retention</text><text x="190" y="92.0" text-anchor="end" font-size="11" fill="#333">Independent problem solving</text><rect x="318.5" y="83.8" width="120.5" height="9.4" fill="#A85B53"/><text x="378.8" y="92.0" text-anchor="middle" font-size="9" fill="#fff">-1</text><text x="190" y="109.0" text-anchor="end" font-size="11" fill="#333">Completion time</text><rect x="439.0" y="100.8" width="241.0" height="9.4" fill="#5E8A6A"/><text x="559.5" y="109.0" text-anchor="middle" font-size="9" fill="#fff">2</text><text x="190" y="126.0" text-anchor="end" font-size="11" fill="#333">Code quality</text><text x="190" y="143.0" text-anchor="end" font-size="11" fill="#333">Assignment score</text><rect x="439.0" y="134.8" width="241.0" height="9.4" fill="#5E8A6A"/><text x="559.5" y="143.0" text-anchor="middle" font-size="9" fill="#fff">2</text><text x="190" y="160.0" text-anchor="end" font-size="11" fill="#333">Metacognition</text><rect x="439.0" y="151.8" width="120.5" height="9.4" fill="#5E8A6A"/><text x="499.2" y="160.0" text-anchor="middle" font-size="9" fill="#fff">1</text><text x="190" y="177.0" text-anchor="end" font-size="11" fill="#333">Over-reliance</text><rect x="318.5" y="168.8" width="120.5" height="9.4" fill="#A85B53"/><text x="378.8" y="177.0" text-anchor="middle" font-size="9" fill="#fff">-1</text><rect x="439.0" y="206.1" width="8.0" height="6" fill="#C99A4A"/><rect x="439.0" y="214.9" width="8.0" height="6" fill="#C99A4A"/><rect x="439.0" y="232.4" width="8.0" height="6" fill="#C99A4A"/><rect x="198" y="14" width="10" height="10" fill="#5E8A6A"/><text x="212" y="23" font-size="10" fill="#333">Positive effect</text><rect x="395" y="14" width="10" height="10" fill="#A85B53"/><text x="409" y="23" font-size="10" fill="#333">Negative effect</text><rect x="592" y="14" width="10" height="10" fill="#C99A4A"/><text x="606" y="23" font-size="10" fill="#333">Null effect</text></svg><p class="chart-interpretation"><strong>What this means: </strong>Positive / negative / null effect-direction evidence counts per outcome. This visual encodes effect_direction, not whether evidence supports a claim.</p></div></div></section><section class="brief-block brief-tribunal" id="brief-tribunal-en"><header class="brief-block-header"><h2>Evidence tribunal</h2><p>Supported, uncertain, contradicted and missing evidence stay separated instead of flattened into long prose.</p></header><div class="brief-block-body"><div class="evidence-tribunal" data-visual="evidence-tribunal-grid"><p class="tribunal-summary"><strong>Decision: </strong>Pilot · <strong>Confidence: </strong>Moderate</p><div class="tribunal-grid"><article class="tribunal-card supported"><header><span aria-hidden="true">✓</span><h3>Can claim</h3><span class="tribunal-count">7</span></header><ul><li><p>AI coding assistants reliably increase task performance during training (completion speed, correctness) — E-001, E-006.</p><div class="tribunal-evidence-refs"><code>E-001</code><code>E-006</code></div></li><li><p>Unguarded generative AI access can harm independent problem solving when access is removed — E-004.</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li><li><p>Guardrail design (hints instead of answers) substantially mitigates the negative learning effect — E-005.</p><div class="tribunal-evidence-refs"><code>E-005</code></div></li></ul><details class="tribunal-more"><summary>View 4 more</summary><ul><li><p>Task performance gains do not automatically imply learning gains — E-004 vs E-006 (within-study contrast).</p><div class="tribunal-evidence-refs"><code>E-004</code><code>E-006</code></div></li><li><p>Tool capability is substantial: Codex solves roughly half to three-quarters of CS1 exam-style questions — E-010.</p><div class="tribunal-evidence-refs"><code>E-010</code></div></li><li><p>Professional-developer RCT shows ~55% faster task completion with Copilot; directness limited by professional population — E-008.</p><div class="tribunal-evidence-refs"><code>E-008</code></div></li><li><p>LLM code explanations rate comparable to student-authored explanations, viable as scaffold material — E-011.</p><div class="tribunal-evidence-refs"><code>E-011</code></div></li></ul></details></article><article class="tribunal-card uncertain"><header><span aria-hidden="true">?</span><h3>Cannot yet claim</h3><span class="tribunal-count">4</span></header><ul><li><p>Whether AI coding assistants improve or preserve actual programming learning in university novices — no direct university-level RCT in reviewed set [无直接证据]</p></li><li><p>Whether one-week neutral retention (Kazemitabaar 2023) extends to a semester — E-003.</p><div class="tribunal-evidence-refs"><code>E-003</code></div></li><li><p>Whether benchmark quality findings (E-009) and explanation-quality ratings (E-011) translate into classroom learning gains.</p><div class="tribunal-evidence-refs"><code>E-009</code><code>E-011</code></div></li></ul><details class="tribunal-more"><summary>View 1 more</summary><ul><li><p>How comprehension/ownership difficulties documented in usability studies (E-012) behave over a full semester with guardrails.</p><div class="tribunal-evidence-refs"><code>E-012</code></div></li></ul></details></article><article class="tribunal-card contradicted"><header><span aria-hidden="true">×</span><h3>Contradicted claims</h3><span class="tribunal-count">2</span></header><ul><li><p>The claim &#x27;AI tools always improve learning&#x27; is contradicted by E-004 (unguarded access, -17% independent exam).</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li><li><p>The claim &#x27;speed gains equal learning gains&#x27; is contradicted by the task-vs-learning separation across E-001/E-006/E-008 vs E-004.</p><div class="tribunal-evidence-refs"><code>E-001</code><code>E-006</code><code>E-008</code><code>E-004</code></div></li></ul></article><article class="tribunal-card missing"><header><span aria-hidden="true">…</span><h3>Missing evidence</h3><span class="tribunal-count">4</span></header><ul><li><p>RCT of AI coding assistants in university programming courses with retention and no-AI transfer tests.</p></li><li><p>Studies varying AI usage policy within the same course.</p></li><li><p>Longitudinal data on AI dependency beyond one course.</p></li></ul><details class="tribunal-more"><summary>View 1 more</summary><ul><li><p>Peer-reviewed replication of the professional speed RCT (Peng et al. remains a preprint).</p></li></ul></details></article></div></div></div></section><section class="brief-block brief-action" id="brief-action-en"><header class="brief-block-header"><h2>Evidence to action</h2><p>Applicability, guardrails, stop conditions and evaluation form one executable path.</p></header><div class="brief-block-body"><div class="evidence-to-action" data-visual="evidence-to-action"><article class="action-node action-evidence"><span>Evidence</span><p class="action-node-text">AI coding assistants reliably increase task performance during training (completion speed, correctness) — E-001, E-006.</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-applicability"><span>Applicability</span><p class="action-node-text">pilot in first-year C course with guardrailed usage policy</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-decision"><span>Decision</span><p class="action-node-text">Pilot</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-guardrails"><span>Guardrails</span><div class="action-node-text"><p>AI usage is allowed in three explicitly graded modes (explain / collaborate / no-AI-transfer).…</p><details class="detail-expander"><summary>Expand full explanation</summary><div class="detail-body"><p>AI usage is allowed in three explicitly graded modes (explain / collaborate / no-AI-transfer). Copying unexamined AI output is an academic integrity violation and is assessed via the reasoning-trace requirement.</p></div></details></div></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-stop"><span>Stop conditions</span><div class="action-node-text"><p>transfer-test scores drop significantly below baseline cohort expectations; widespread integrity violations in reasoning traces;…</p><details class="detail-expander"><summary>Expand full explanation</summary><div class="detail-body"><p>transfer-test scores drop significantly below baseline cohort expectations; widespread integrity violations in reasoning traces; AI dependency signals exceed threshold in risk metrics; TA/teacher workload becomes unsustainable</p></div></details></div></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-evaluation"><span>Evaluation</span><div class="action-node-text"><p>treatment group shows non-inferior independent problem solving (delta within 5%) AND superior or equal retention AND AI dependency index below thresho…</p><details class="detail-expander"><summary>Expand full explanation</summary><div class="detail-body"><p>treatment group shows non-inferior independent problem solving (delta within 5%) AND superior or equal retention AND AI dependency index below threshold; if independent problem solving declines &gt;10%, the pilot is judged unsuccessful regardless of task-performance gains.</p></div></details></div></article></div></div></section><section class="brief-block brief-lieflat" id="brief-lieflat-en"><header class="brief-block-header"><h2>Lieflat Editorial Gallery</h2><p>Charts selected and composed by AI from the Lieflat catalog; every number traces back to result.json.</p></header><div class="brief-block-body"><div class="lieflat-gallery-container"><figure class="lieflat-card" data-lieflat data-visual="lieflat-bubble_almanac" data-chart-id="lieflat-bubble-almanac.svg"><h3 class="lieflat-title">Year × dimension evidence almanac</h3><p class="lieflat-sub">Bubble area ∝ study count · solid core = significant results</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 540 300" width="100%" height="100%" role="img" aria-label="Year × dimension evidence almanac" style="background:#111827;">
2252
+ <line x1="44" y1="70.0" x2="520" y2="70.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:0ms"/>
2253
+ <line x1="44" y1="77.0" x2="520" y2="77.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:14ms"/>
2254
+ <line x1="44" y1="84.0" x2="520" y2="84.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:28ms"/>
2255
+ <line x1="44" y1="91.0" x2="520" y2="91.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:42ms"/>
2256
+ <line x1="44" y1="98.0" x2="520" y2="98.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:56ms"/>
2257
+ <line x1="44" y1="105.0" x2="520" y2="105.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:70ms"/>
2258
+ <line x1="44" y1="112.0" x2="520" y2="112.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:84ms"/>
2259
+ <line x1="44" y1="119.0" x2="520" y2="119.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:98ms"/>
2260
+ <line x1="44" y1="126.0" x2="520" y2="126.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:112ms"/>
2261
+ <line x1="44" y1="133.0" x2="520" y2="133.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:126ms"/>
2262
+ <line x1="44" y1="140.0" x2="520" y2="140.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:140ms"/>
2263
+ <line x1="44" y1="147.0" x2="520" y2="147.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:154ms"/>
2264
+ <line x1="44" y1="154.0" x2="520" y2="154.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:168ms"/>
2265
+ <line x1="44" y1="161.0" x2="520" y2="161.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:182ms"/>
2266
+ <line x1="44" y1="168.0" x2="520" y2="168.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:196ms"/>
2267
+ <line x1="44" y1="175.0" x2="520" y2="175.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:210ms"/>
2268
+ <line x1="44" y1="182.0" x2="520" y2="182.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:224ms"/>
2269
+ <line x1="44" y1="189.0" x2="520" y2="189.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:238ms"/>
2270
+ <line x1="44" y1="196.0" x2="520" y2="196.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:252ms"/>
2271
+ <line x1="44" y1="203.0" x2="520" y2="203.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:266ms"/>
2272
+ <line x1="44" y1="210.0" x2="520" y2="210.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:280ms"/>
2273
+ <line x1="44" y1="217.0" x2="520" y2="217.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:294ms"/>
2274
+ <line x1="44" y1="224.0" x2="520" y2="224.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:308ms"/>
2275
+ <line x1="44" y1="231.0" x2="520" y2="231.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:322ms"/>
2276
+ <line x1="44" y1="238.0" x2="520" y2="238.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:336ms"/>
2277
+ <line x1="44" y1="245.0" x2="520" y2="245.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:350ms"/>
2278
+ <line x1="44" y1="252.0" x2="520" y2="252.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:364ms"/>
2279
+ <line x1="44" y1="259.0" x2="520" y2="259.0" stroke="#1E293B" stroke-width="0.5" class="lf-fade" style="--motion-delay:378ms"/>
2280
+ <text x="150" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:40ms">Completion tim</text>
2281
+ <text x="203" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:140ms">Independent pr</text>
2282
+ <text x="256" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:240ms">Retention</text>
2283
+ <text x="309" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:340ms">Assignment sco</text>
2284
+ <text x="361" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:440ms">Knowledge gain</text>
2285
+ <text x="414" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:540ms">Code quality</text>
2286
+ <text x="467" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:640ms">Metacognition</text>
2287
+ <text x="488.1" y="76" fill="#F8FAFC" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:740ms">Over-reliance</text>
2288
+ <text x="96" y="96.0" fill="#94A3B8" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:80ms">2022</text>
2289
+ <text x="96" y="140.0" fill="#94A3B8" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:180ms">2023</text>
2290
+ <text x="96" y="184.0" fill="#94A3B8" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:280ms">2024</text>
2291
+ <text x="96" y="228.0" fill="#94A3B8" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:380ms">2025</text>
2292
+ <circle cx="308.6" cy="92.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:100ms"><title>Assignment score (2022) — N = 1 studies, significant = 0</title></circle>
2293
+ <circle cx="520.0" cy="92.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:112ms"><title>Over-reliance (2022) — N = 1 studies, significant = 0</title></circle>
2294
+ <circle cx="150.0" cy="136.0" r="5.1" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:124ms"><title>Completion time (2023) — N = 2 studies, significant = 0</title></circle>
2295
+ <circle cx="202.9" cy="136.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:136ms"><title>Independent problem solving (2023) — N = 1 studies, significant = 0</title></circle>
2296
+ <circle cx="255.7" cy="136.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:148ms"><title>Retention (2023) — N = 1 studies, significant = 0</title></circle>
2297
+ <circle cx="414.3" cy="136.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:160ms"><title>Code quality (2023) — N = 1 studies, significant = 0</title></circle>
2298
+ <circle cx="467.1" cy="136.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:172ms"><title>Metacognition (2023) — N = 1 studies, significant = 0</title></circle>
2299
+ <circle cx="361.4" cy="180.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:184ms"><title>Knowledge gain (2024) — N = 1 studies, significant = 0</title></circle>
2300
+ <circle cx="202.9" cy="224.0" r="5.1" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:196ms"><title>Independent problem solving (2025) — N = 2 studies, significant = 0</title></circle>
2301
+ <circle cx="308.6" cy="224.0" r="3.6" fill="#38BDF8" fill-opacity="0.22" stroke="#38BDF8" stroke-width="1.2" class="lf-pop" style="--motion-delay:208ms"><title>Assignment score (2025) — N = 1 studies, significant = 0</title></circle>
2302
+ </svg></div><figcaption class="lieflat-caption">Drawn only when years and outcome dimensions exist.</figcaption><p class="lieflat-src">L9 Bubble Almanac · Evidence.year X Dimension</p></figure><figure class="lieflat-card" data-lieflat data-visual="lieflat-matrix_heat" data-chart-id="lieflat-matrix-heat.svg"><h3 class="lieflat-title">Year × outcome evidence density</h3><p class="lieflat-sub">Each cell counts evidence items for that year and outcome</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 540 440" width="100%" height="100%" role="img" aria-label="Year × outcome evidence density" style="background:#111827;">
2303
+ <text x="196.2" y="66" text-anchor="middle" font-size="9" font-weight="800" fill="#94A3B8" class="lf-fade" style="--motion-delay:40ms">2022</text>
2304
+ <text x="288.8" y="66" text-anchor="middle" font-size="9" font-weight="800" fill="#94A3B8" class="lf-fade" style="--motion-delay:140ms">2023</text>
2305
+ <text x="381.2" y="66" text-anchor="middle" font-size="9" font-weight="800" fill="#94A3B8" class="lf-fade" style="--motion-delay:240ms">2024</text>
2306
+ <text x="473.8" y="66" text-anchor="middle" font-size="9" font-weight="800" fill="#94A3B8" class="lf-fade" style="--motion-delay:340ms">2025</text>
2307
+ <text x="138" y="97.0" text-anchor="end" font-size="8" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:60ms">Completion tim</text>
2308
+ <rect x="152.0" y="80.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:0ms"/>
2309
+ <text x="196.2" y="97.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:60ms">0</text>
2310
+ <rect x="244.5" y="80.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.90" class="lf-pop" style="--motion-delay:12ms"/>
2311
+ <text x="288.8" y="97.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:72ms">2</text>
2312
+ <rect x="337.0" y="80.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:24ms"/>
2313
+ <text x="381.2" y="97.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:84ms">0</text>
2314
+ <rect x="429.5" y="80.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:36ms"/>
2315
+ <text x="473.8" y="97.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:96ms">0</text>
2316
+ <text x="138" y="129.0" text-anchor="end" font-size="8" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:160ms">Independent pr</text>
2317
+ <rect x="152.0" y="112.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:12ms"/>
2318
+ <text x="196.2" y="129.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:72ms">0</text>
2319
+ <rect x="244.5" y="112.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.54" class="lf-pop" style="--motion-delay:24ms"/>
2320
+ <text x="288.8" y="129.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:84ms">1</text>
2321
+ <rect x="337.0" y="112.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:36ms"/>
2322
+ <text x="381.2" y="129.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:96ms">0</text>
2323
+ <rect x="429.5" y="112.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.90" class="lf-pop" style="--motion-delay:48ms"/>
2324
+ <text x="473.8" y="129.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:108ms">2</text>
2325
+ <text x="138" y="161.0" text-anchor="end" font-size="8" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:260ms">Retention</text>
2326
+ <rect x="152.0" y="144.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:24ms"/>
2327
+ <text x="196.2" y="161.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:84ms">0</text>
2328
+ <rect x="244.5" y="144.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.54" class="lf-pop" style="--motion-delay:36ms"/>
2329
+ <text x="288.8" y="161.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:96ms">1</text>
2330
+ <rect x="337.0" y="144.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:48ms"/>
2331
+ <text x="381.2" y="161.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:108ms">0</text>
2332
+ <rect x="429.5" y="144.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:60ms"/>
2333
+ <text x="473.8" y="161.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:120ms">0</text>
2334
+ <text x="138" y="193.0" text-anchor="end" font-size="8" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:360ms">Assignment sco</text>
2335
+ <rect x="152.0" y="176.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.54" class="lf-pop" style="--motion-delay:36ms"/>
2336
+ <text x="196.2" y="193.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:96ms">1</text>
2337
+ <rect x="244.5" y="176.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:48ms"/>
2338
+ <text x="288.8" y="193.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:108ms">0</text>
2339
+ <rect x="337.0" y="176.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:60ms"/>
2340
+ <text x="381.2" y="193.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:120ms">0</text>
2341
+ <rect x="429.5" y="176.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.54" class="lf-pop" style="--motion-delay:72ms"/>
2342
+ <text x="473.8" y="193.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:132ms">1</text>
2343
+ <text x="138" y="225.0" text-anchor="end" font-size="8" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:460ms">Knowledge gain</text>
2344
+ <rect x="152.0" y="208.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:48ms"/>
2345
+ <text x="196.2" y="225.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:108ms">0</text>
2346
+ <rect x="244.5" y="208.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:60ms"/>
2347
+ <text x="288.8" y="225.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:120ms">0</text>
2348
+ <rect x="337.0" y="208.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.54" class="lf-pop" style="--motion-delay:72ms"/>
2349
+ <text x="381.2" y="225.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:132ms">1</text>
2350
+ <rect x="429.5" y="208.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:84ms"/>
2351
+ <text x="473.8" y="225.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:144ms">0</text>
2352
+ <text x="138" y="257.0" text-anchor="end" font-size="8" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:560ms">Code quality</text>
2353
+ <rect x="152.0" y="240.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:60ms"/>
2354
+ <text x="196.2" y="257.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:120ms">0</text>
2355
+ <rect x="244.5" y="240.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.54" class="lf-pop" style="--motion-delay:72ms"/>
2356
+ <text x="288.8" y="257.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:132ms">1</text>
2357
+ <rect x="337.0" y="240.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:84ms"/>
2358
+ <text x="381.2" y="257.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:144ms">0</text>
2359
+ <rect x="429.5" y="240.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:96ms"/>
2360
+ <text x="473.8" y="257.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:156ms">0</text>
2361
+ <text x="138" y="289.0" text-anchor="end" font-size="8" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:660ms">Metacognition</text>
2362
+ <rect x="152.0" y="272.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:72ms"/>
2363
+ <text x="196.2" y="289.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:132ms">0</text>
2364
+ <rect x="244.5" y="272.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.54" class="lf-pop" style="--motion-delay:84ms"/>
2365
+ <text x="288.8" y="289.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:144ms">1</text>
2366
+ <rect x="337.0" y="272.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:96ms"/>
2367
+ <text x="381.2" y="289.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:156ms">0</text>
2368
+ <rect x="429.5" y="272.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:108ms"/>
2369
+ <text x="473.8" y="289.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:168ms">0</text>
2370
+ <text x="138" y="321.0" text-anchor="end" font-size="8" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:760ms">Over-reliance</text>
2371
+ <rect x="152.0" y="304.0" width="88.5" height="28" rx="4" fill="#38BDF8" fill-opacity="0.54" class="lf-pop" style="--motion-delay:84ms"/>
2372
+ <text x="196.2" y="321.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:144ms">1</text>
2373
+ <rect x="244.5" y="304.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:96ms"/>
2374
+ <text x="288.8" y="321.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:156ms">0</text>
2375
+ <rect x="337.0" y="304.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:108ms"/>
2376
+ <text x="381.2" y="321.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:168ms">0</text>
2377
+ <rect x="429.5" y="304.0" width="88.5" height="28" rx="4" fill="#1E293B" fill-opacity="0.5" class="lf-fade" style="--motion-delay:120ms"/>
2378
+ <text x="473.8" y="321.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-pop" style="--motion-delay:180ms">0</text>
2379
+ </svg></div><figcaption class="lieflat-caption">Shows where the evidence sits across years and outcomes.</figcaption><p class="lieflat-src">L16 Matrix Heat · Evidence.year X Outcome Counts</p></figure><figure class="lieflat-card" data-lieflat data-visual="lieflat-tick_rows" data-chart-id="lieflat-tick-rows.svg"><h3 class="lieflat-title">Effect direction by outcome</h3><p class="lieflat-sub">One dot = one evidence item · green = positive · grey = null · orange = negative</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 540 300" width="100%" height="100%" role="img" aria-label="Effect direction by outcome" style="background:#111827;">
2143
2380
  <text x="128" y="102.0" text-anchor="end" font-size="9" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:40ms">Completion t</text>
2144
2381
  <circle cx="140.0" cy="99.0" r="2.3" fill="#10B981" class="lf-pop" style="--motion-delay:0ms"><title>Completion time — positive evidence</title></circle>
2145
2382
  <circle cx="148.0" cy="99.0" r="2.3" fill="#10B981" class="lf-pop" style="--motion-delay:12ms"><title>Completion time — positive evidence</title></circle>
@@ -2168,7 +2405,90 @@ select:focus-visible,
2168
2405
  <text x="128" y="256.0" text-anchor="end" font-size="9" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:740ms">Over-relianc</text>
2169
2406
  <circle cx="140.0" cy="253.0" r="2.3" fill="#38BDF8" class="lf-pop" style="--motion-delay:700ms"><title>Over-reliance — negative evidence</title></circle>
2170
2407
  <text x="512" y="256.0" text-anchor="end" font-size="8.0" font-weight="800" fill="#F8FAFC" class="lf-fade" style="--motion-delay:1200ms">-1</text>
2171
- </svg></div><figcaption class="lieflat-caption">Based on effect_direction counts; all numbers come from result.json.</figcaption><p class="lieflat-src">F5 Tick Rows · outcomes.direction_counts</p></figure><div class="lieflat-suppressed" role="note"><strong>2 charts suppressed (insufficient data; mirrors the Meaningful Visualization Gate)</strong><ul><li><code>FOREST-PLOT (publication figure)</code>: fewer than 3 studies with numeric effect size (got 0)</li><li><code>L2 Dot Cascade</code>: fewer than 3 studies with numeric effect size (got 0)</li></ul></div></div></div></section><section class="brief-block brief-sources" id="brief-sources-en"><header class="brief-block-header"><h2>Key sources</h2><p>Only the key sources in the brief; full traceability expands in the full report.</p></header><div class="brief-block-body"><div class="brief-source-grid"><article class="brief-source"><code>S-2023-kazemitabaar</code><h3><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</a></h3><p>Tier 1 DOI-verified paper · 2023</p></article><article class="brief-source"><code>S-2025-bastani</code><h3><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">Generative AI without guardrails can harm learning: Evidence from high school mathematics</a></h3><p>Tier 1 DOI-verified paper · 2025</p></article><article class="brief-source"><code>S-2024-marzuki</code><h3><a href="https://link.springer.com/article/10.1186/s40561-024-00295-9">Impact of ChatGPT on ESL students&#x27; academic writing skills</a></h3><p>Tier 1 DOI-verified paper · 2024</p></article><article class="brief-source"><code>S-2023-peng</code><h3><a href="https://doi.org/10.48550/arXiv.2302.06590">The Impact of AI on Developer Productivity: Evidence from GitHub Copilot</a></h3><p>tier2_academic_database · 2023</p></article></div><p class="brief-source-more">4 more sources are traceable in the full report.</p></div></section></div></div>
2408
+ </svg></div><figcaption class="lieflat-caption">Based on effect_direction counts; all numbers come from result.json.</figcaption><p class="lieflat-src">F5 Tick Rows · Outcomes.direction Counts</p></figure><figure class="lieflat-card" data-lieflat data-visual="lieflat-paired_rungs" data-chart-id="lieflat-paired-rungs.svg"><h3 class="lieflat-title">Positive vs negative evidence by outcome</h3><p class="lieflat-sub">Two columns summarise positive and negative evidence counts</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 554 300" width="100%" height="100%" role="img" aria-label="Positive vs negative evidence by outcome" style="background:#111827;">
2409
+ <text x="60" y="76" font-size="8" font-weight="700" fill="#10B981" class="lf-fade" style="--motion-delay:40ms">POSITIVE</text>
2410
+ <text x="60" y="92" font-size="8" font-weight="700" fill="#38BDF8" class="lf-fade" style="--motion-delay:80ms">NEGATIVE</text>
2411
+ <rect x="73.0" y="230.3" width="13" height="5.5" rx="2.6" fill="#10B981" class="lf-fade" style="--motion-delay:60ms"><title>Knowledge gain — positive</title></rect>
2412
+ <text x="90.0" y="254" text-anchor="middle" font-size="8.0" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:100ms">Knowledge </text>
2413
+ <text x="90.0" y="222.3" text-anchor="middle" font-size="8" font-weight="800" fill="#F8FAFC" class="lf-fade" style="--motion-delay:520ms">1 / 0</text>
2414
+ <text x="150.0" y="254" text-anchor="middle" font-size="8.0" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:200ms">Retention</text>
2415
+ <text x="150.0" y="230.0" text-anchor="middle" font-size="8" font-weight="800" fill="#F8FAFC" class="lf-fade" style="--motion-delay:620ms">0 / 0</text>
2416
+ <rect x="214.0" y="230.3" width="13" height="5.5" rx="2.6" fill="#38BDF8" class="lf-fade" style="--motion-delay:260ms"><title>Independent problem solving — negative</title></rect>
2417
+ <text x="210.0" y="254" text-anchor="middle" font-size="8.0" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:300ms">Independen</text>
2418
+ <text x="210.0" y="222.3" text-anchor="middle" font-size="8" font-weight="800" fill="#F8FAFC" class="lf-fade" style="--motion-delay:720ms">0 / 1</text>
2419
+ <rect x="253.0" y="230.3" width="13" height="5.5" rx="2.6" fill="#10B981" class="lf-fade" style="--motion-delay:360ms"><title>Completion time — positive</title></rect>
2420
+ <rect x="253.0" y="222.6" width="13" height="5.5" rx="2.6" fill="#10B981" class="lf-fade" style="--motion-delay:372ms"><title>Completion time — positive</title></rect>
2421
+ <text x="270.0" y="254" text-anchor="middle" font-size="8.0" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:400ms">Completion</text>
2422
+ <text x="270.0" y="214.6" text-anchor="middle" font-size="8" font-weight="800" fill="#F8FAFC" class="lf-fade" style="--motion-delay:820ms">2 / 0</text>
2423
+ <text x="330.0" y="254" text-anchor="middle" font-size="8.0" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:500ms">Code quali</text>
2424
+ <text x="330.0" y="230.0" text-anchor="middle" font-size="8" font-weight="800" fill="#F8FAFC" class="lf-fade" style="--motion-delay:920ms">0 / 0</text>
2425
+ <rect x="373.0" y="230.3" width="13" height="5.5" rx="2.6" fill="#10B981" class="lf-fade" style="--motion-delay:560ms"><title>Assignment score — positive</title></rect>
2426
+ <rect x="373.0" y="222.6" width="13" height="5.5" rx="2.6" fill="#10B981" class="lf-fade" style="--motion-delay:572ms"><title>Assignment score — positive</title></rect>
2427
+ <text x="390.0" y="254" text-anchor="middle" font-size="8.0" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:600ms">Assignment</text>
2428
+ <text x="390.0" y="214.6" text-anchor="middle" font-size="8" font-weight="800" fill="#F8FAFC" class="lf-fade" style="--motion-delay:1020ms">2 / 0</text>
2429
+ <rect x="433.0" y="230.3" width="13" height="5.5" rx="2.6" fill="#10B981" class="lf-fade" style="--motion-delay:660ms"><title>Metacognition — positive</title></rect>
2430
+ <text x="450.0" y="254" text-anchor="middle" font-size="8.0" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:700ms">Metacognit</text>
2431
+ <text x="450.0" y="222.3" text-anchor="middle" font-size="8" font-weight="800" fill="#F8FAFC" class="lf-fade" style="--motion-delay:1120ms">1 / 0</text>
2432
+ <rect x="514.0" y="230.3" width="13" height="5.5" rx="2.6" fill="#38BDF8" class="lf-fade" style="--motion-delay:760ms"><title>Over-reliance — negative</title></rect>
2433
+ <text x="510.0" y="254" text-anchor="middle" font-size="8.0" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:800ms">Over-relia</text>
2434
+ <text x="510.0" y="222.3" text-anchor="middle" font-size="8" font-weight="800" fill="#F8FAFC" class="lf-fade" style="--motion-delay:1220ms">0 / 1</text>
2435
+ <line x1="46" y1="238" x2="512" y2="238" stroke="#F8FAFC" stroke-width="1.2" class="lf-draw" style="--motion-delay:120ms"/>
2436
+ </svg></div><figcaption class="lieflat-caption">Splits positive and negative evidence so they never cancel out.</figcaption><p class="lieflat-src">F6 Paired Rungs · Outcomes.paired Counts</p></figure><figure class="lieflat-card" data-lieflat data-visual="lieflat-brand_spectrum" data-chart-id="lieflat-brand-spectrum.svg"><h3 class="lieflat-title">Net effect direction by outcome</h3><p class="lieflat-sub">Position = (positive - negative) / directional count · centre is neutral</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 540 360" width="100%" height="100%" role="img" aria-label="Net effect direction by outcome" style="background:#111827;">
2437
+ <text x="138" y="62" text-anchor="end" font-size="9" font-weight="700" fill="#94A3B8" class="lf-fade" style="--motion-delay:40ms">Negative-led</text>
2438
+ <text x="412" y="62" font-size="9" font-weight="700" fill="#94A3B8" class="lf-fade" style="--motion-delay:80ms">Positive-led</text>
2439
+ <path d="M 400.0 88 C 400.0 111.0, 150.0 111.0, 150.0 134 C 150.0 157.0, 400.0 157.0, 400.0 180 C 400.0 203.0, 400.0 203.0, 400.0 226 C 400.0 249.0, 400.0 249.0, 400.0 272 C 400.0 295.0, 150.0 295.0, 150.0 318" fill="none" stroke="#1E293B" stroke-width="26" stroke-linecap="round" stroke-linejoin="round" opacity="0.95" class="lf-draw" style="--motion-delay:60ms"/>
2440
+ <line x1="150" y1="88" x2="400" y2="88" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:60ms"/>
2441
+ <line x1="150" y1="84" x2="150" y2="92" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:60ms"/>
2442
+ <line x1="400" y1="84" x2="400" y2="92" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:60ms"/>
2443
+ <text x="134" y="91" text-anchor="end" font-size="8.5" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:90ms">Knowledge </text>
2444
+ <circle cx="400.0" cy="88" r="7.5" fill="#10B981" stroke="#111827" stroke-width="1.8" class="lf-pop" style="--motion-delay:160ms"><title>Knowledge gain — Negative-led↔Positive-led: +100% (pos 1 / neg 0 / null 0)</title></circle>
2445
+ <text x="400.0" y="77" fill="#F8FAFC" font-size="8" font-weight="800" text-anchor="middle" class="lf-fade" style="--motion-delay:220ms">+100%</text>
2446
+ <line x1="150" y1="134" x2="400" y2="134" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:160ms"/>
2447
+ <line x1="150" y1="130" x2="150" y2="138" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:160ms"/>
2448
+ <line x1="400" y1="130" x2="400" y2="138" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:160ms"/>
2449
+ <text x="134" y="137" text-anchor="end" font-size="8.5" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:190ms">Independen</text>
2450
+ <circle cx="150.0" cy="134" r="7.5" fill="#38BDF8" stroke="#111827" stroke-width="1.8" class="lf-pop" style="--motion-delay:260ms"><title>Independent problem solving — Negative-led↔Positive-led: -100% (pos 0 / neg 1 / null 2)</title></circle>
2451
+ <text x="150.0" y="123" fill="#F8FAFC" font-size="8" font-weight="800" text-anchor="middle" class="lf-fade" style="--motion-delay:320ms">-100%</text>
2452
+ <line x1="150" y1="180" x2="400" y2="180" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:260ms"/>
2453
+ <line x1="150" y1="176" x2="150" y2="184" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:260ms"/>
2454
+ <line x1="400" y1="176" x2="400" y2="184" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:260ms"/>
2455
+ <text x="134" y="183" text-anchor="end" font-size="8.5" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:290ms">Completion</text>
2456
+ <circle cx="400.0" cy="180" r="7.5" fill="#10B981" stroke="#111827" stroke-width="1.8" class="lf-pop" style="--motion-delay:360ms"><title>Completion time — Negative-led↔Positive-led: +100% (pos 2 / neg 0 / null 0)</title></circle>
2457
+ <text x="400.0" y="169" fill="#F8FAFC" font-size="8" font-weight="800" text-anchor="middle" class="lf-fade" style="--motion-delay:420ms">+100%</text>
2458
+ <line x1="150" y1="226" x2="400" y2="226" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:360ms"/>
2459
+ <line x1="150" y1="222" x2="150" y2="230" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:360ms"/>
2460
+ <line x1="400" y1="222" x2="400" y2="230" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:360ms"/>
2461
+ <text x="134" y="229" text-anchor="end" font-size="8.5" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:390ms">Assignment</text>
2462
+ <circle cx="400.0" cy="226" r="7.5" fill="#10B981" stroke="#111827" stroke-width="1.8" class="lf-pop" style="--motion-delay:460ms"><title>Assignment score — Negative-led↔Positive-led: +100% (pos 2 / neg 0 / null 0)</title></circle>
2463
+ <text x="400.0" y="215" fill="#F8FAFC" font-size="8" font-weight="800" text-anchor="middle" class="lf-fade" style="--motion-delay:520ms">+100%</text>
2464
+ <line x1="150" y1="272" x2="400" y2="272" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:460ms"/>
2465
+ <line x1="150" y1="268" x2="150" y2="276" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:460ms"/>
2466
+ <line x1="400" y1="268" x2="400" y2="276" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:460ms"/>
2467
+ <text x="134" y="275" text-anchor="end" font-size="8.5" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:490ms">Metacognit</text>
2468
+ <circle cx="400.0" cy="272" r="7.5" fill="#10B981" stroke="#111827" stroke-width="1.8" class="lf-pop" style="--motion-delay:560ms"><title>Metacognition — Negative-led↔Positive-led: +100% (pos 1 / neg 0 / null 0)</title></circle>
2469
+ <text x="400.0" y="261" fill="#F8FAFC" font-size="8" font-weight="800" text-anchor="middle" class="lf-fade" style="--motion-delay:620ms">+100%</text>
2470
+ <line x1="150" y1="318" x2="400" y2="318" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:560ms"/>
2471
+ <line x1="150" y1="314" x2="150" y2="322" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:560ms"/>
2472
+ <line x1="400" y1="314" x2="400" y2="322" stroke="#334155" stroke-width="1" class="lf-fade" style="--motion-delay:560ms"/>
2473
+ <text x="134" y="321" text-anchor="end" font-size="8.5" font-weight="600" fill="#94A3B8" class="lf-fade" style="--motion-delay:590ms">Over-relia</text>
2474
+ <circle cx="150.0" cy="318" r="7.5" fill="#38BDF8" stroke="#111827" stroke-width="1.8" class="lf-pop" style="--motion-delay:660ms"><title>Over-reliance — Negative-led↔Positive-led: -100% (pos 0 / neg 1 / null 0)</title></circle>
2475
+ <text x="150.0" y="307" fill="#F8FAFC" font-size="8" font-weight="800" text-anchor="middle" class="lf-fade" style="--motion-delay:720ms">-100%</text>
2476
+ <text x="30" y="338" font-size="8" font-weight="600" fill="#64748B" class="lf-fade" style="--motion-delay:400ms">position = (positive − negative) ÷ total direction counts</text>
2477
+ </svg></div><figcaption class="lieflat-caption">Bipolar view of whether each outcome leans supportive or against.</figcaption><p class="lieflat-src">L7 Brand Spectrum · Outcomes.bipolar Axes</p></figure><figure class="lieflat-card" data-lieflat data-visual="lieflat-hundred_field" data-chart-id="lieflat-hundred-field.svg"><h3 class="lieflat-title">Study-design composition</h3><p class="lieflat-sub">One cell = one study · shows which designs produced the evidence</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 540 300" width="100%" height="100%" role="img" aria-label="Study-design composition" style="background:#111827;">
2478
+ <rect x="40.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#38BDF8" class="lf-pop" style="--motion-delay:0ms"><title>rct — 1 study</title></rect>
2479
+ <rect x="58.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#38BDF8" class="lf-pop" style="--motion-delay:12ms"><title>rct — 1 study</title></rect>
2480
+ <rect x="76.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#38BDF8" class="lf-pop" style="--motion-delay:24ms"><title>rct — 1 study</title></rect>
2481
+ <rect x="94.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#38BDF8" class="lf-pop" style="--motion-delay:36ms"><title>rct — 1 study</title></rect>
2482
+ <rect x="112.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#38BDF8" class="lf-pop" style="--motion-delay:48ms"><title>rct — 1 study</title></rect>
2483
+ <rect x="130.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#38BDF8" class="lf-pop" style="--motion-delay:60ms"><title>rct — 1 study</title></rect>
2484
+ <rect x="148.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#38BDF8" class="lf-pop" style="--motion-delay:72ms"><title>rct — 1 study</title></rect>
2485
+ <rect x="166.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#10B981" class="lf-pop" style="--motion-delay:84ms"><title>observational — 1 study</title></rect>
2486
+ <rect x="184.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#10B981" class="lf-pop" style="--motion-delay:96ms"><title>observational — 1 study</title></rect>
2487
+ <rect x="202.0" y="74.0" width="15.5" height="15.5" rx="3" fill="#10B981" class="lf-pop" style="--motion-delay:108ms"><title>observational — 1 study</title></rect>
2488
+ <rect x="40.0" y="92.0" width="15.5" height="15.5" rx="3" fill="#F59E0B" class="lf-pop" style="--motion-delay:120ms"><title>mixed_methods — 1 study</title></rect>
2489
+ <rect x="58.0" y="92.0" width="15.5" height="15.5" rx="3" fill="#64748B" class="lf-pop" style="--motion-delay:132ms"><title>qualitative — 1 study</title></rect>
2490
+ <text x="30" y="270" font-size="8" font-weight="600" fill="#64748B" class="lf-fade" style="--motion-delay:400ms">one cell = one study</text>
2491
+ </svg></div><figcaption class="lieflat-caption">Reveals design skew faster than a table when several designs are present.</figcaption><p class="lieflat-src">L14 Hundred Field · Evidence.study Type Composition</p></figure></div></div></section><section class="brief-block brief-sources" id="brief-sources-en"><header class="brief-block-header"><h2>Key sources</h2><p>Only the key sources in the brief; full traceability expands in the full report.</p></header><div class="brief-block-body"><div class="brief-source-grid"><article class="brief-source"><code>S-2023-kazemitabaar</code><h3><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</a></h3><p>Tier 1 DOI-verified paper · 2023</p></article><article class="brief-source"><code>S-2025-bastani</code><h3><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">Generative AI without guardrails can harm learning: Evidence from high school mathematics</a></h3><p>Tier 1 DOI-verified paper · 2025</p></article><article class="brief-source"><code>S-2024-marzuki</code><h3><a href="https://link.springer.com/article/10.1186/s40561-024-00295-9">Impact of ChatGPT on ESL students&#x27; academic writing skills</a></h3><p>Tier 1 DOI-verified paper · 2024</p></article><article class="brief-source"><code>S-2023-peng</code><h3><a href="https://doi.org/10.48550/arXiv.2302.06590">The Impact of AI on Developer Productivity: Evidence from GitHub Copilot</a></h3><p>Tier2 Academic Database · 2023</p></article></div><p class="brief-source-more">4 more sources are traceable in the full report.</p></div></section></div></div>
2172
2492
  <div class="report-page report-page-full" data-report-page="full" hidden>
2173
2493
  <div class="full-report-intro"><h2>Full Report</h2><p>Conclusions first: every traceable piece of evidence and method note lives here, with visuals only at points where they add meaning. Every number traces back to result.json.</p></div>
2174
2494
  <div class="full-report-layout"><aside class="full-report-toc" aria-label="Contents"><div class="toc-head"><strong>Contents</strong><button type="button" class="toc-collapse" aria-expanded="true" data-label-collapse="Collapse contents" data-label-expand="Expand contents">Collapse contents</button></div><nav><a href="#full-01-decision-en" data-toc-target="full-01-decision-en" data-chapter-key="decision">01 Decision, Adjudication &amp; Research Boundary</a><a href="#full-02-evidence-en" data-toc-target="full-02-evidence-en" data-chapter-key="evidence">02 Key Evidence &amp; Outcome Separation</a><a href="#full-03-quality-en" data-toc-target="full-03-quality-en" data-chapter-key="quality">03 Evidence Quality, Counterevidence &amp; Method Audit</a><a href="#full-04-action-en" data-toc-target="full-04-action-en" data-chapter-key="action">04 Applicability &amp; Teaching Action</a><a href="#full-05-evaluation-en" data-toc-target="full-05-evaluation-en" data-chapter-key="evaluation">05 Pilot, Evaluation &amp; Stop Conditions</a><a href="#full-06-sources-en" data-toc-target="full-06-sources-en" data-chapter-key="sources">06 Sources, Traceability &amp; Appendix</a></nav></aside><main class="full-report-content"><section id="full-01-decision-en" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>01 Decision, Adjudication &amp; Research Boundary</h2><p class="full-chapter-lead">State the final adjudication and research boundary before explaining why.</p></header><div class="full-chapter-body">
@@ -2180,13 +2500,13 @@ select:focus-visible,
2180
2500
  </div>
2181
2501
  <p class="hero-rationale">Positive task-performance evidence plus documented unguarded-access risk, mixed quality/usability signals, and missing university-level learning evidence → bounded, guardrailed pilot with evaluation, not full adoption.</p>
2182
2502
  <div class="hero-insights">
2183
- <article class="hero-insight support"><span>Strongest supported conclusion</span><p class="hero-insight-text">AI coding assistants raise task performance for novices during training.</p></article>
2184
- <article class="hero-insight uncertain"><span>Key uncertainty / contradiction</span><p class="hero-insight-text">Whether AI coding assistants improve or preserve actual programming learning in university novices — no direct university-level RCT in reviewed set [无直接证据]</p></article>
2185
- <article class="hero-insight risk"><span>Main risk</span><p class="hero-insight-text">AI dependency and over-reliance risk is real and documented for unguarded usage (E-004) and foreshadowed by usability findings (E-012).</p></article>
2186
- <article class="hero-insight next"><span>Next action</span><p class="hero-insight-text">Not provided in this research result.</p></article>
2503
+ <article class="hero-insight support"><span>Strongest supported conclusion</span><p class="hero-insight-text">AI coding assistants reliably speed up practice work: completion rate 1.15x and time 0.57x in a randomised trial of 69 novices.</p></article>
2504
+ <article class="hero-insight uncertain"><span>Key uncertainty / contradiction</span><p class="hero-insight-text">No university-level RCT measures learning directly, and the one large trial that did - unguarded GPT-4 - saw independent exam scores fall 17%.</p></article>
2505
+ <article class="hero-insight risk"><span>Main risk</span><p class="hero-insight-text">Unguarded access can raise practice performance while lowering independent exam performance, and learners may not notice the gap.</p></article>
2506
+ <article class="hero-insight next"><span>Next action</span><p class="hero-insight-text">Run a phased CS1 pilot with hints-not-answers guardrails, weekly lab use, and a no-AI transfer exam that can stop the pilot.</p></article>
2187
2507
  </div>
2188
2508
  <p class="hero-provenance"><span>Evidence / sources</span> · 12 / 8</p>
2189
- </div><div class="scope-grid"><article class="scope-card"><h3>Research question</h3><p class="scope-text">我们准备在大学一年级 C 语言课程中允许学生使用生成式 AI 编程助手。它到底会不会提高学习效果?应该怎样引入?</p></article><article class="scope-card"><h3>Target learners</h3><p class="scope-text">Education level: First-year undergraduate; Major: Computer science; Prior knowledge: first_programming_course_no_prior_text_based_programming; Learner characteristics: mixed_ability_large_class_60_students</p></article><article class="scope-card"><h3>Course context</h3><p class="scope-text">Subject: C programming; Course type: lecture_lab; Duration: 16_weeks_one_semester</p></article><article class="scope-card"><h3>AI intervention</h3><p class="scope-text">teaching_method: lecture_with_lab_exercises; AI tool: generative_ai_coding_assistant; Allowed usage: under_design_pending_evidence_review; Frequency: weekly_lab_sessions; Duration: one_semester</p></article><article class="scope-card"><h3>Comparison</h3><p class="scope-text">no_ai_coding_assistant_control</p></article><article class="scope-card"><h3>Outcome constructs</h3><p class="scope-text">Primary outcomes: Independent problem solving, Code quality; Secondary outcomes: Completion time, Retention, Knowledge gain; Risk outcomes: AI dependency, Over-reliance, Reduced transfer</p></article><article class="scope-card"><h3>Research scope</h3><p class="scope-text">Time range: 2021-2026; Geography: worldwide; Study designs: Randomized controlled trial, Quasi-experimental, Observational</p></article><article class="scope-card"><h3>Decision success condition</h3><p class="scope-text">independent problem solving and code quality improve (or do not decline) while AI dependency risk stays controlled; evidence base supports a bounded pilot.</p></article></div></div></section><section id="full-02-evidence-en" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>02 Key Evidence &amp; Outcome Separation</h2><p class="full-chapter-lead">Place task performance, actual learning, retention and risk on one evidence map without conflating them.</p></header><div class="full-chapter-body"><div class="retrieval-grid"><article><h3>Inclusion criteria</h3><ul><li>studies_of_generative_AI_coding_tools_in_learning_to_program</li><li>outcomes_measuring_learning_not_only_task_speed</li><li>university_or_novice_programming_populations</li></ul></article><article><h3>Exclusion criteria</h3><ul><li>practitioner_anecdotes_without_data</li><li>industry_professional_populations_only</li></ul></article></div><div class="retrieval-coverage"><h3>Source coverage</h3><p><code>S-2023-kazemitabaar</code> <code>S-2025-bastani</code> <code>S-2024-marzuki</code> <code>S-2023-peng</code> <code>S-2023-yetistiren</code> <code>S-2022-finnie-ansley</code> <code>S-2023-explanations-compare</code> <code>S-2022-vaithilingam</code></p><p class="retrieval-note">This report shows only retrieval metadata present in result; it does not fabricate PRISMA/funnel counts when none exist.</p></div><div class="outcome-separation" data-visual="outcome-separation"><div class="visual-heading"><h3>Outcome Separation · Task performance ≠ learning</h3><p>Outcomes are adjudicated separately so faster training performance is not silently treated as evidence of learning.</p><p class="semantic-note">Effect direction comes from evidence.effect_direction; supporting a claim does not imply a positive outcome.</p></div><div class="outcome-groups"><article class="outcome-group outcome-task"><h3>Task / proximal performance</h3><ul><li><strong>Completion time</strong><span class="outcome-states"><span class="dir pos">Positive effect 2</span></span></li><li><strong>Code quality</strong><span class="outcome-states"><span class="dir neu">Null effect 1</span></span></li><li><strong>Assignment score</strong><span class="outcome-states"><span class="dir pos">Positive effect 2</span></span></li></ul></article><article class="outcome-group outcome-learning"><h3>Learning / retention / transfer</h3><ul><li><strong>Knowledge gain</strong><span class="outcome-states"><span class="dir pos">Positive effect 1</span></span></li><li><strong>Retention</strong><span class="outcome-states"><span class="dir neu">Null effect 1</span></span></li><li><strong>Independent problem solving</strong><span class="outcome-states"><span class="dir neg">Negative effect 1</span><span class="dir neu">Null effect 2</span></span></li></ul></article><article class="outcome-group outcome-risk"><h3>Risk / dependency</h3><ul><li><strong>Over-reliance</strong><span class="outcome-states"><span class="dir neg">Negative effect 1</span></span></li></ul></article><article class="outcome-group outcome-other"><h3>Other outcomes</h3><ul><li><strong>Metacognition</strong><span class="outcome-states"><span class="dir pos">Positive effect 1</span></span></li></ul></article></div></div><div class='table-wrap outcome-table'><table class='data-table'><thead><tr><th>Outcome</th><th>Positive effect</th><th>Negative effect</th><th>Null effect</th><th>Evidence</th></tr></thead><tbody><tr><td><strong>Knowledge gain</strong><span class='raw-tag' title='raw id'>knowledge_gain</span></td><td class='num'>1</td><td class='num'>0</td><td class='num'>0</td><td><code>E-007</code> </td></tr><tr><td><strong>Retention</strong><span class='raw-tag' title='raw id'>retention</span></td><td class='num'>0</td><td class='num'>0</td><td class='num'>1</td><td><code>E-003</code> </td></tr><tr><td><strong>Independent problem solving</strong><span class='raw-tag' title='raw id'>independent_problem_solving</span></td><td class='num'>0</td><td class='num'>1</td><td class='num'>2</td><td><code>E-002</code> <code>E-004</code> <code>E-005</code> </td></tr><tr><td><strong>Completion time</strong><span class='raw-tag' title='raw id'>completion_time</span></td><td class='num'>2</td><td class='num'>0</td><td class='num'>0</td><td><code>E-001</code> <code>E-008</code> </td></tr><tr><td><strong>Code quality</strong><span class='raw-tag' title='raw id'>code_quality</span></td><td class='num'>0</td><td class='num'>0</td><td class='num'>1</td><td><code>E-009</code> </td></tr><tr><td><strong>Assignment score</strong><span class='raw-tag' title='raw id'>assignment_score</span></td><td class='num'>2</td><td class='num'>0</td><td class='num'>0</td><td><code>E-006</code> <code>E-010</code> </td></tr><tr><td><strong>Metacognition</strong><span class='raw-tag' title='raw id'>metacognition</span></td><td class='num'>1</td><td class='num'>0</td><td class='num'>0</td><td><code>E-011</code> </td></tr><tr><td><strong>Over-reliance</strong><span class='raw-tag' title='raw id'>over_reliance</span></td><td class='num'>0</td><td class='num'>1</td><td class='num'>0</td><td><code>E-012</code> </td></tr></tbody></table></div><div class="visual-surface" data-visual="outcome-evidence-balance"><svg viewBox="0 0 720 300" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="Outcome evidence effect balance"><title>Outcome evidence effect balance</title><desc>Positive / negative / null effect counts per outcome (based on effect_direction, not claim support).</desc><rect x="0" y="0" width="720" height="300" fill="#FFFFFF"/><line x1="415.0" y1="46" x2="415.0" y2="182" stroke="#999" stroke-width="1" stroke-dasharray="3,3"/><text x="142" y="58.0" text-anchor="end" font-size="11" fill="#333">Knowledge gain</text><rect x="415.0" y="49.8" width="132.5" height="9.4" fill="#5E8A6A"/><text x="481.2" y="58.0" text-anchor="middle" font-size="9" fill="#fff">1</text><text x="142" y="75.0" text-anchor="end" font-size="11" fill="#333">Retention</text><text x="142" y="92.0" text-anchor="end" font-size="11" fill="#333">Independent problem solving</text><rect x="282.5" y="83.8" width="132.5" height="9.4" fill="#A85B53"/><text x="348.8" y="92.0" text-anchor="middle" font-size="9" fill="#fff">-1</text><text x="142" y="109.0" text-anchor="end" font-size="11" fill="#333">Completion time</text><rect x="415.0" y="100.8" width="265.0" height="9.4" fill="#5E8A6A"/><text x="547.5" y="109.0" text-anchor="middle" font-size="9" fill="#fff">2</text><text x="142" y="126.0" text-anchor="end" font-size="11" fill="#333">Code quality</text><text x="142" y="143.0" text-anchor="end" font-size="11" fill="#333">Assignment score</text><rect x="415.0" y="134.8" width="265.0" height="9.4" fill="#5E8A6A"/><text x="547.5" y="143.0" text-anchor="middle" font-size="9" fill="#fff">2</text><text x="142" y="160.0" text-anchor="end" font-size="11" fill="#333">Metacognition</text><rect x="415.0" y="151.8" width="132.5" height="9.4" fill="#5E8A6A"/><text x="481.2" y="160.0" text-anchor="middle" font-size="9" fill="#fff">1</text><text x="142" y="177.0" text-anchor="end" font-size="11" fill="#333">Over-reliance</text><rect x="282.5" y="168.8" width="132.5" height="9.4" fill="#A85B53"/><text x="348.8" y="177.0" text-anchor="middle" font-size="9" fill="#fff">-1</text><rect x="415.0" y="206.1" width="8.0" height="6" fill="#C99A4A"/><rect x="415.0" y="214.9" width="8.0" height="6" fill="#C99A4A"/><rect x="415.0" y="232.4" width="8.0" height="6" fill="#C99A4A"/><rect x="150" y="14" width="10" height="10" fill="#5E8A6A"/><text x="164" y="23" font-size="10" fill="#333">Positive effect</text><rect x="347" y="14" width="10" height="10" fill="#A85B53"/><text x="361" y="23" font-size="10" fill="#333">Negative effect</text><rect x="544" y="14" width="10" height="10" fill="#C99A4A"/><text x="558" y="23" font-size="10" fill="#333">Null effect</text></svg><p class="chart-interpretation"><strong>What this means: </strong>Positive / negative / null effect-direction evidence counts per outcome. This visual encodes effect_direction, not whether evidence supports a claim.</p></div><div id="chart-outcome-en" class="chart-mount" aria-label="Outcome Evidence Overview"></div><figure class="academic-figure" data-visual="outcome-evidence-balance"><svg viewBox="0 0 720 300" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="Effect direction by outcome type (publication figure)"><title>Effect direction by outcome type (publication figure)</title><desc>Counts of positive / negative / null effects per outcome type with an integer count axis, theme-independent. Source: EduEvidence result.json.</desc><rect width="720" height="300" fill="#FFFFFF"/><line x1="70" y1="250" x2="650" y2="250" stroke="#333" stroke-width="1"/><text x="106.2" y="266" text-anchor="middle" font-size="10" fill="#333">knowledge_gain</text><rect x="88.1" y="150.0" width="18.1" height="100.0" fill="#38BDF8"/><text x="97.2" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><rect x="106.2" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="124.4" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><text x="178.8" y="266" text-anchor="middle" font-size="10" fill="#333">retention</text><rect x="160.6" y="250.0" width="18.1" height="1.0" fill="#38BDF8"/><rect x="178.8" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="196.9" y="150.0" width="18.1" height="100.0" fill="#F59E0B"/><text x="205.9" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><text x="251.2" y="266" text-anchor="middle" font-size="10" fill="#333">independent_problem_solving</text><rect x="233.1" y="250.0" width="18.1" height="1.0" fill="#38BDF8"/><rect x="251.2" y="150.0" width="18.1" height="100.0" fill="#10B981"/><text x="260.3" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><rect x="269.4" y="50.0" width="18.1" height="200.0" fill="#F59E0B"/><text x="278.4" y="47.0" text-anchor="middle" font-size="9" fill="#333">2</text><text x="323.8" y="266" text-anchor="middle" font-size="10" fill="#333">completion_time</text><rect x="305.6" y="50.0" width="18.1" height="200.0" fill="#38BDF8"/><text x="314.7" y="47.0" text-anchor="middle" font-size="9" fill="#333">2</text><rect x="323.8" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="341.9" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><text x="396.2" y="266" text-anchor="middle" font-size="10" fill="#333">code_quality</text><rect x="378.1" y="250.0" width="18.1" height="1.0" fill="#38BDF8"/><rect x="396.2" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="414.4" y="150.0" width="18.1" height="100.0" fill="#F59E0B"/><text x="423.4" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><text x="468.8" y="266" text-anchor="middle" font-size="10" fill="#333">assignment_score</text><rect x="450.6" y="50.0" width="18.1" height="200.0" fill="#38BDF8"/><text x="459.7" y="47.0" text-anchor="middle" font-size="9" fill="#333">2</text><rect x="468.8" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="486.9" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><text x="541.2" y="266" text-anchor="middle" font-size="10" fill="#333">metacognition</text><rect x="523.1" y="150.0" width="18.1" height="100.0" fill="#38BDF8"/><text x="532.2" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><rect x="541.2" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="559.4" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><text x="613.8" y="266" text-anchor="middle" font-size="10" fill="#333">over_reliance</text><rect x="595.6" y="250.0" width="18.1" height="1.0" fill="#38BDF8"/><rect x="613.8" y="150.0" width="18.1" height="100.0" fill="#10B981"/><text x="622.8" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><rect x="631.9" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><line x1="65" y1="250.0" x2="70" y2="250.0" stroke="#999"/><text x="62" y="253.0" text-anchor="end" font-size="9" fill="#666">0</text><line x1="65" y1="150.0" x2="70" y2="150.0" stroke="#999"/><text x="62" y="153.0" text-anchor="end" font-size="9" fill="#666">1</text><line x1="65" y1="50.0" x2="70" y2="50.0" stroke="#999"/><text x="62" y="53.0" text-anchor="end" font-size="9" fill="#666">2</text><text x="360.0" y="30" text-anchor="middle" font-size="14" font-weight="700" fill="#111">Effect direction by outcome type</text><rect x="70" y="8" width="10" height="10" fill="#38BDF8"/><text x="84" y="17" font-size="10" fill="#333">Positive effect</text><rect x="267" y="8" width="10" height="10" fill="#10B981"/><text x="281" y="17" font-size="10" fill="#333">Negative effect</text><rect x="464" y="8" width="10" height="10" fill="#F59E0B"/><text x="478" y="17" font-size="10" fill="#333">Null effect</text><text x="20" y="290" font-size="11" fill="#333333" font-style="italic">Fig. 1. Counts of positive / negative / null effects per outcome type (based on effect_direction; publication figure, theme-independent). Source: EduEvidence result.json.</text></svg><figcaption>Fig. 1. Positive / negative / null effect counts per outcome (based on effect_direction, not claim support).</figcaption></figure><div class='matrix-controls'><div class='matrix-tools'><input id='matrix-search-full-en' type='search' placeholder='Search evidence…' aria-label='Filter / search evidence'><select id='matrix-direction-full-en' aria-label='Filter by effect direction'><option value=''>All effects</option><option value='positive'>Positive effect</option><option value='negative'>Negative effect</option><option value='null'>Null effect</option></select><select id='matrix-outcome-full-en' aria-label='Filter by outcome type'><option value=''>All outcomes</option><option value='assignment_score'>Assignment score</option><option value='code_quality'>Code quality</option><option value='completion_time'>Completion time</option><option value='independent_problem_solving'>Independent problem solving</option><option value='knowledge_gain'>Knowledge gain</option><option value='metacognition'>Metacognition</option><option value='over_reliance'>Over-reliance</option><option value='retention'>Retention</option></select></div></div><div class='table-wrap matrix-wrap'><table id='evidence-matrix-full-en' class='data-table evidence-matrix'><thead><tr><th>ID</th><th>Outcome</th><th>Effect</th><th>Quality</th><th>Claim</th><th>Source</th></tr></thead><tbody><tr data-effect="positive" data-direction="positive" data-outcome="completion_time" data-search="e-001 study-kazemitabaar-2023 smpl-kazemitabaar-2023-n69 studying the effect of ai code generators on supporting novice learners in introductory programming ai coding assistants significantly increase task completion speed and completion rate during training. 69 novices ages 10-17 with no prior text-based programming experience access_to_openai_codex_ai_coding_assistant_during_training baseline_group_without_ai_coding_assistant positive s-2023-kazemitabaar"><td><code>E-001</code></td><td><strong>Completion time</strong></td><td><span class="dir pos">Positive effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">9</span><span class="quality-meter" aria-hidden="true"><i style="width:90%"></i></span></div></td><td class="claim-cell"><p>AI coding assistants significantly increase task completion speed and completion rate during training.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-KAZEMITABAAR-2023</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-KAZEMITABAAR-2023-N69</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Randomized controlled trial</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>k12_ages_10_17</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>69 novices ages 10-17 with no prior text-based programming experience</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>69</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>access_to_openai_codex_ai_coding_assistant_during_training</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>baseline_group_without_ai_coding_assistant</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>code_authoring_task_progress_and_time</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>1.15x completion rate, 0.57x time, 1.8x correctness</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Positive effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>3_weeks_training</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>controlled experiment with random assignment, immediate post-test and 1-week retention test</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>randomized_controlled_design;immediate_post_test_and_retention_test;code_modification_task_guard</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>non_university_population_ages_10_17;small_sample_69;self-paced environment differs from classroom</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>prior_programming_competency_interaction</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1_study_design=2 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=2 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>9.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=partial_novice_programmers_but_younger · subject_match=introductory_programming · tool_match=codex_like_generative_ai · scope=task_performance_during_training</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.7</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>AI coding assistants significantly increase task completion speed and completion rate during training.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3544548.3580919</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">https://dl.acm.org/doi/10.1145/3544548.3580919</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3544548.3580919" title="Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming"><code>S-2023-kazemitabaar</code></a></td></tr><tr data-effect="null" data-direction="null" data-outcome="independent_problem_solving" data-search="e-002 study-kazemitabaar-2023 smpl-kazemitabaar-2023-n69 studying the effect of ai code generators on supporting novice learners in introductory programming access to ai code generation did not decrease performance on manual code-modification tasks. 69 novices ages 10-17 with no prior text-based programming experience access_to_openai_codex_ai_coding_assistant_during_training baseline_group_without_ai_coding_assistant null s-2023-kazemitabaar"><td><code>E-002</code></td><td><strong>Independent problem solving</strong></td><td><span class="dir neu">Null effect</span><span class="relation-note">Claim relation: Neutral</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>Access to AI code generation did not decrease performance on manual code-modification tasks.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-KAZEMITABAAR-2023</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-KAZEMITABAAR-2023-N69</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Randomized controlled trial</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>k12_ages_10_17</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>69 novices ages 10-17 with no prior text-based programming experience</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>69</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>access_to_openai_codex_ai_coding_assistant_during_training</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>baseline_group_without_ai_coding_assistant</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>manual code-modification tasks during training</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>no significant difference between groups</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Null effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Neutral</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>3_weeks_training</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>controlled experiment, code-modification task followed each code-authoring task</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>direct_test_of_transfer-adjacent_skill;same_session_measurement</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>code modification is not full independent problem solving;non_university population</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>practice_effect</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1_study_design=2 · D2_sample_quality=2 · D3_measurement_validity=1 · D4_temporal_strength=1 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=partial · subject_match=introductory_programming · tool_match=codex · scope=short-term manual code modification</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.5</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Access to AI code generation did not decrease performance on manual code-modification tasks.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3544548.3580919</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">https://dl.acm.org/doi/10.1145/3544548.3580919</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3544548.3580919" title="Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming"><code>S-2023-kazemitabaar</code></a></td></tr><tr data-effect="null" data-direction="null" data-outcome="retention" data-search="e-003 study-kazemitabaar-2023 smpl-kazemitabaar-2023-n69 studying the effect of ai code generators on supporting novice learners in introductory programming one week after training, retention differences between codex and baseline groups did not reach statistical significance. 69 novices ages 10-17 with no prior text-based programming experience access_to_openai_codex_ai_coding_assistant_during_training baseline_group_without_ai_coding_assistant null s-2023-kazemitabaar"><td><code>E-003</code></td><td><strong>Retention</strong></td><td><span class="dir neu">Null effect</span><span class="relation-note">Claim relation: Neutral</span></td><td><div class="quality-cell"><span class="num">9</span><span class="quality-meter" aria-hidden="true"><i style="width:90%"></i></span></div></td><td class="claim-cell"><p>One week after training, retention differences between Codex and baseline groups did not reach statistical significance.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-KAZEMITABAAR-2023</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-KAZEMITABAAR-2023-N69</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Randomized controlled trial</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>k12_ages_10_17</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>69 novices ages 10-17 with no prior text-based programming experience</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>69</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>access_to_openai_codex_ai_coding_assistant_during_training</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>baseline_group_without_ai_coding_assistant</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>retention post-test one week after training</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>slightly better for Codex group but not significant</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Null effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Neutral</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>3_weeks_training_plus_1_week_retention</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>controlled experiment with delayed retention test</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>delayed_test_included</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>1-week retention window is short;small sample;non-university population</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>prior_competency</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1_study_design=2 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=2 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>9.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=partial · subject_match=introductory_programming · tool_match=codex · scope=retention over one week</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.5</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>One week after training, retention differences between Codex and baseline groups did not reach statistical significance.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3544548.3580919</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">https://dl.acm.org/doi/10.1145/3544548.3580919</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3544548.3580919" title="Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming"><code>S-2023-kazemitabaar</code></a></td></tr><tr data-effect="negative" data-direction="negative" data-outcome="independent_problem_solving" data-search="e-004 study-bastani-2025 smpl-bastani-2025-n950 generative ai without guardrails can harm learning: evidence from high school mathematics students with unguarded gpt-4 access performed 17% worse on the independent exam than the control group, despite higher practice performance. nearly a thousand high school math students in turkey gpt4_based_tutor_gpt_base_unguarded no_generative_ai_control negative s-2025-bastani"><td><code>E-004</code></td><td><strong>Independent problem solving</strong></td><td><span class="dir neg">Negative effect</span><span class="relation-note">Claim relation: Contradict</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>Students with unguarded GPT-4 access performed 17% worse on the independent exam than the control group, despite higher practice performance.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-BASTANI-2025</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-BASTANI-2025-N950</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Randomized controlled trial</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>high_school</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>nearly a thousand high school math students in Turkey</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>950</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>gpt4_based_tutor_gpt_base_unguarded</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>no_generative_ai_control</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>exam without access to AI resources after practice phase</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>negative_17_percent_on_independent_exam</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Negative effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Contradict</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>in_class_study_sessions</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>large-scale randomized controlled trial, practice phase then closed-book exam</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>large_scale_rct;independent_exam_without_ai;arm_wise_design</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>high_school_mathematics_not_university_programming;single_country</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>tool_design_difference</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1_study_design=2 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=1 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=partial_same_age_band_different_subject · subject_match=no_mathematics_vs_programming · tool_match=gpt4_chat_interface · scope=unguarded_general_chat_interface</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.75</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Students with unguarded GPT-4 access performed 17% worse on the independent exam than the control group, despite higher practice performance.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://www.pnas.org/doi/10.1073/pnas.2422633122</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">https://www.pnas.org/doi/10.1073/pnas.2422633122</a></dd></div></dl></details></td><td><a class="source-link" href="https://www.pnas.org/doi/10.1073/pnas.2422633122" title="Generative AI without guardrails can harm learning: Evidence from high school mathematics"><code>S-2025-bastani</code></a></td></tr><tr data-effect="null" data-direction="null" data-outcome="independent_problem_solving" data-search="e-005 study-bastani-2025 smpl-bastani-2025-n950 generative ai without guardrails can harm learning: evidence from high school mathematics guardrail design of the ai tutor (hints instead of answers, teacher-informed prompts) largely eliminated the negative learning effect. nearly a thousand high school math students in turkey gpt4_tutor_with_teacher_designed_guardrails no_generative_ai_control null s-2025-bastani"><td><code>E-005</code></td><td><strong>Independent problem solving</strong></td><td><span class="dir neu">Null effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>Guardrail design of the AI tutor (hints instead of answers, teacher-informed prompts) largely eliminated the negative learning effect.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-BASTANI-2025</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-BASTANI-2025-N950</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Randomized controlled trial</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>high_school</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>nearly a thousand high school math students in Turkey</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>950</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>gpt4_tutor_with_teacher_designed_guardrails</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>no_generative_ai_control</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>exam without access to AI resources after practice phase</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>negative effect essentially eradicated, no positive effect observed</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Null effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>in_class_study_sessions</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>large-scale randomized controlled trial, three arms</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>direct_manipulation_of_tool_design;large_sample</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>no_positive_learning_gain_even_with_guardrails;subject_mismatch</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>prompt_engineering_effort</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1_study_design=2 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=1 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=partial · subject_match=no · tool_match=guardrailed_tutor_design · scope=guardrail_design_principle_transferable</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.75</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Guardrail design of the AI tutor (hints instead of answers, teacher-informed prompts) largely eliminated the negative learning effect.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://www.pnas.org/doi/10.1073/pnas.2422633122</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">https://www.pnas.org/doi/10.1073/pnas.2422633122</a></dd></div></dl></details></td><td><a class="source-link" href="https://www.pnas.org/doi/10.1073/pnas.2422633122" title="Generative AI without guardrails can harm learning: Evidence from high school mathematics"><code>S-2025-bastani</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="assignment_score" data-search="e-006 study-bastani-2025 smpl-bastani-2025-n950 generative ai without guardrails can harm learning: evidence from high school mathematics access to gpt-4 during practice improves task performance (48% for gpt base, 127% for gpt tutor) — but this task performance does not transfer to independent exam performance. nearly a thousand high school math students in turkey gpt4_tutor_access_during_practice no_generative_ai_control positive s-2025-bastani"><td><code>E-006</code></td><td><strong>Assignment score</strong></td><td><span class="dir pos">Positive effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>Access to GPT-4 during practice improves task performance (48% for GPT Base, 127% for GPT Tutor) — but this task performance does not transfer to independent exam performance.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-BASTANI-2025</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-BASTANI-2025-N950</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Randomized controlled trial</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>high_school</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>nearly a thousand high school math students in Turkey</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>950</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>gpt4_tutor_access_during_practice</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>no_generative_ai_control</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>practice problem performance during study sessions</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>48-127 percent improvement on practice problems</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Positive effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>in_class_study_sessions</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>randomized controlled trial with practice and closed-book exam phases</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>same_study_compares_task_and_learning;large_sample</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>subject_mismatch_mathematics</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>task_familiarity</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1_study_design=2 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=1 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=partial · subject_match=no · tool_match=gpt4 · scope=task_performance_vs_learning_separation</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.75</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Access to GPT-4 during practice improves task performance (48% for GPT Base, 127% for GPT Tutor) — but this task performance does not transfer to independent exam performance.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://www.pnas.org/doi/10.1073/pnas.2422633122</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">https://www.pnas.org/doi/10.1073/pnas.2422633122</a></dd></div></dl></details></td><td><a class="source-link" href="https://www.pnas.org/doi/10.1073/pnas.2422633122" title="Generative AI without guardrails can harm learning: Evidence from high school mathematics"><code>S-2025-bastani</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="knowledge_gain" data-search="e-007 study-marzuki-2024 smpl-marzuki-2024-n72 impact of chatgpt on esl students&#x27; academic writing skills chatgpt as a formative feedback tool produced a significant positive impact on students&#x27; academic writing skills with positive student perceptions. undergraduate esl students at an indian university chatgpt_as_formative_feedback_tool traditional_instruction_control positive s-2024-marzuki"><td><code>E-007</code></td><td><strong>Knowledge gain</strong></td><td><span class="dir pos">Positive effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">6</span><span class="quality-meter" aria-hidden="true"><i style="width:60%"></i></span></div></td><td class="claim-cell"><p>ChatGPT as a formative feedback tool produced a significant positive impact on students&#x27; academic writing skills with positive student perceptions.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-MARZUKI-2024</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-MARZUKI-2024-N72</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Impact of ChatGPT on ESL students&#x27; academic writing skills</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Impact of ChatGPT on ESL students&#x27; academic writing skills</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2024</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Mixed methods</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>undergraduate</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>undergraduate ESL students at an Indian university</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>72</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>chatgpt_as_formative_feedback_tool</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>traditional_instruction_control</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>writing tests with pre-post-delayed design</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>significant positive impact on writing skills</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Positive effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>6_hours_intervention</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>mixed methods intervention study, pre/post/delayed tests and focus groups</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>delayed_post_test;mixed_methods_triangulation</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>short_intervention_6_hours;single_institution;elite_private_university</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>self_selection_consent</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1_study_design=1 · D2_sample_quality=1 · D3_measurement_validity=2 · D4_temporal_strength=2 · D5_directness=0</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>6.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=yes_undergraduate · subject_match=no_writing_not_programming · tool_match=chatgpt · scope=formative_feedback_writing</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>ChatGPT as a formative feedback tool produced a significant positive impact on students&#x27; academic writing skills with positive student perceptions.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://link.springer.com/article/10.1186/s40561-024-00295-9</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://link.springer.com/article/10.1186/s40561-024-00295-9">https://link.springer.com/article/10.1186/s40561-024-00295-9</a></dd></div></dl></details></td><td><a class="source-link" href="https://link.springer.com/article/10.1186/s40561-024-00295-9" title="Impact of ChatGPT on ESL students&#x27; academic writing skills"><code>S-2024-marzuki</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="completion_time" data-search="e-008 study-peng-2023 smpl-peng-2023-n95 the impact of ai on developer productivity: evidence from github copilot professional developers with copilot access completed a standardized coding task about 55% faster than the control group (rct, n=95). 95 recruited professional developers completing a standardized coding task on a freelance platform access_to_github_copilot_during_task control_group_without_copilot positive s-2023-peng"><td><code>E-008</code></td><td><strong>Completion time</strong></td><td><span class="dir pos">Positive effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>Professional developers with Copilot access completed a standardized coding task about 55% faster than the control group (RCT, n=95).</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-PENG-2023</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-PENG-2023-N95</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>The Impact of AI on Developer Productivity: Evidence from GitHub Copilot</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>The Impact of AI on Developer Productivity: Evidence from GitHub Copilot</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Randomized controlled trial</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>professional_developers_not_students</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>95 recruited professional developers completing a standardized coding task on a freelance platform</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>95</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>access_to_github_copilot_during_task</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>control_group_without_copilot</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>time_to_complete_http_server_implementation</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>~55.8% faster task completion in Copilot group</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Positive effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>single_task_session</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>online randomized controlled experiment with objective completion-time metric</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>randomized_controlled_design;objective_completion_time_metric</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>professional_population_not_students;single_task_ecology;preprint_not_peer_reviewed</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>task_familiarity;platform_recruitment_self_selection</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1_study_design=2 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=1 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=mismatch_professional_developers · subject_match=adjacent_web_development_task · tool_match=copilot_like_generative_ai · scope=task_performance_only_no_learning_outcome</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.6</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Professional developers with Copilot access completed a standardized coding task about 55% faster than the control group (RCT, n=95).</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://doi.org/10.48550/arXiv.2302.06590</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://doi.org/10.48550/arXiv.2302.06590">https://doi.org/10.48550/arXiv.2302.06590</a></dd></div></dl></details></td><td><a class="source-link" href="https://doi.org/10.48550/arXiv.2302.06590" title="The Impact of AI on Developer Productivity: Evidence from GitHub Copilot"><code>S-2023-peng</code></a></td></tr><tr data-effect="null" data-direction="null" data-outcome="code_quality" data-search="e-009 study-yetistiren-2023 smpl-yetistiren-2023-bench github copilot ai pair programmer: asset or liability? systematic benchmark evaluation reports mixed quality results for copilot-generated code relative to human code: correctness competitive on parts of the benchmark while security-relevant defects are documented. copilot-generated and human-written programs drawn from published benchmark datasets copilot_generated_programs human_written_programs_on_same_benchmarks null s-2023-yetistiren"><td><code>E-009</code></td><td><strong>Code quality</strong></td><td><span class="dir neu">Null effect</span><span class="relation-note">Claim relation: Neutral</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>Systematic benchmark evaluation reports mixed quality results for Copilot-generated code relative to human code: correctness competitive on parts of the benchmark while security-relevant defects are documented.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-YETISTIREN-2023</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-YETISTIREN-2023-BENCH</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>GitHub Copilot AI Pair Programmer: Asset or Liability?</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>GitHub Copilot AI Pair Programmer: Asset or Liability?</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Observational</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>not_applicable_code_artifacts</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>Copilot-generated and human-written programs drawn from published benchmark datasets</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>copilot_generated_programs</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>human_written_programs_on_same_benchmarks</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>correctness_security_maintainability_metrics_on_benchmarks</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>mixed quality profile; no single-direction summary</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Null effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Neutral</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>not_applicable_artifact_study</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>systematic empirical evaluation of generated code against human baselines on public benchmarks</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>multi_dimensional_quality_metrics;reproducible_benchmark_protocol</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>artifact_benchmark_not_classroom;no_learning_outcome;tool_version_from_2023</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>benchmark_task_distribution</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1_study_design=1 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=1 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=mismatch_no_learners_in_study · subject_match=introductory_adjacent_code_tasks · tool_match=copilot_like_generative_ai · scope=output_quality_only</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Systematic benchmark evaluation reports mixed quality results for Copilot-generated code relative to human code: correctness competitive on parts of the benchmark while security-relevant defects are documented.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://doi.org/10.1016/j.jss.2023.111734</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://doi.org/10.1016/j.jss.2023.111734">https://doi.org/10.1016/j.jss.2023.111734</a></dd></div></dl></details></td><td><a class="source-link" href="https://doi.org/10.1016/j.jss.2023.111734" title="GitHub Copilot AI Pair Programmer: Asset or Liability?"><code>S-2023-yetistiren</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="assignment_score" data-search="e-010 study-finnieansley-2022 smpl-finnieansley-2022-qsets using github copilot to solve introductory programming problems codex produced passing-level solutions for roughly half to three-quarters of cs1 exam-style questions depending on the dataset, indicating substantial task-capability headroom available to novices. cs1 exam-style question sets answered by codex and compared against published student score distributions codex_answer_generation_on_cs1_questions published_student_cohort_score_distributions positive s-2022-finnie-ansley"><td><code>E-010</code></td><td><strong>Assignment score</strong></td><td><span class="dir pos">Positive effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>Codex produced passing-level solutions for roughly half to three-quarters of CS1 exam-style questions depending on the dataset, indicating substantial task-capability headroom available to novices.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-FINNIEANSLEY-2022</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-FINNIEANSLEY-2022-QSETS</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Using GitHub Copilot to Solve Introductory Programming Problems</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Using GitHub Copilot to Solve Introductory Programming Problems</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2022</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Observational</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>university_year_1_question_sets</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>CS1 exam-style question sets answered by Codex and compared against published student score distributions</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>codex_answer_generation_on_cs1_questions</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>published_student_cohort_score_distributions</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>pass_rate_on_cs1_exam_style_questions</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>passing solutions on ~50-75% of questions across datasets</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Positive effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>not_applicable_capability_probe</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>capability benchmark against published student distributions; reproducible question sets</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>public_reproducible_question_sets;directly_relevant_task_domain</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>tool_solves_task_does_not_equate_student_learning;codex_2021_model_version_outdated</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>question_leakage_into_training_data_possible</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1_study_design=1 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=1 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=partial_measures_tool_not_students · subject_match=introductory_programming · tool_match=copilot_like_generative_ai · scope=tool_capability_headroom</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Codex produced passing-level solutions for roughly half to three-quarters of CS1 exam-style questions depending on the dataset, indicating substantial task-capability headroom available to novices.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3545945.3569830</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3545945.3569830">https://dl.acm.org/doi/10.1145/3545945.3569830</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3545945.3569830" title="Using GitHub Copilot to Solve Introductory Programming Problems"><code>S-2022-finnie-ansley</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="metacognition" data-search="e-011 study-explcomp-2023 smpl-explcomp-2023-ratings comparing code explanations created by students and large language models controlled comparisons find llm-generated code explanations comparable to (in places better than) student-authored explanations, suggesting viability as explanatory scaffold material rather than as a replacement for student explanation practice. student-produced versus llm-produced explanations of short programs under controlled comparison llm_generated_code_explanations student_generated_explanations_of_same_programs positive s-2023-explanations-compare"><td><code>E-011</code></td><td><strong>Metacognition</strong></td><td><span class="dir pos">Positive effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>Controlled comparisons find LLM-generated code explanations comparable to (in places better than) student-authored explanations, suggesting viability as explanatory scaffold material rather than as a replacemen…</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-EXPLCOMP-2023</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-EXPLCOMP-2023-RATINGS</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Comparing Code Explanations Created by Students and Large Language Models</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Comparing Code Explanations Created by Students and Large Language Models</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Observational</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>university_introductory</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>Student-produced versus LLM-produced explanations of short programs under controlled comparison</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>llm_generated_code_explanations</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>student_generated_explanations_of_same_programs</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>rated_explanation_quality_and_comprehensibility</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>comparable-or-better rated quality vs student explanations</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Positive effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>single_session_ratings</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>controlled comparison with blind rating of explanation pairs</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>controlled_pairwise_comparison;learning_process_relevant_construct</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>short_term_ratings_not_learning_gains;small_program_snippets_ecology</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>rating_criteria_subjectivity</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1_study_design=1 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=1 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=partial_scaffold_material_only · subject_match=introductory_programming · tool_match=llm_explanations · scope=scaffold_quality_not_effectiveness</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Controlled comparisons find LLM-generated code explanations comparable to (in places better than) student-authored explanations, suggesting viability as explanatory scaffold material rather than as a replacement for student explanation practice.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3587102.3588785</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3587102.3588785">https://dl.acm.org/doi/10.1145/3587102.3588785</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3587102.3588785" title="Comparing Code Explanations Created by Students and Large Language Models"><code>S-2023-explanations-compare</code></a></td></tr><tr data-effect="negative" data-direction="negative" data-outcome="over_reliance" data-search="e-012 study-vaithilingam-2022 smpl-vaithilingam-2022-n24 expectation vs. experience: evaluating the usability of code generation tools despite faster first-task completion, participants struggled to understand and debug ai-generated solutions and reported low ownership of the final program - documenting metacognitive and dependence risks that pure speed metrics miss. 24 participants in a within-subjects usability study of copilot-style tools copilot_assisted_program_writing within_subject_baseline_without_tool negative s-2022-vaithilingam"><td><code>E-012</code></td><td><strong>Over-reliance</strong></td><td><span class="dir neg">Negative effect</span><span class="relation-note">Claim relation: Contradict</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>Despite faster first-task completion, participants struggled to understand and debug AI-generated solutions and reported low ownership of the final program - documenting metacognitive and dependence risks that…</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-VAITHILINGAM-2022</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-VAITHILINGAM-2022-N24</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Expectation vs. Experience: Evaluating the Usability of Code Generation Tools</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Expectation vs. Experience: Evaluating the Usability of Code Generation Tools</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2022</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Qualitative</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>mixed_cs_students_and_professionals</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>24 participants in a within-subjects usability study of Copilot-style tools</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>24</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>copilot_assisted_program_writing</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>within_subject_baseline_without_tool</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>understanding_ownership_and_debugging_reports</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>documented comprehension/ownership difficulties despite speed gain</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Negative effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Contradict</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>single_session</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>within-subject usability study with tasks, observation and interviews</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>rich_qualitative_process_data;constructs_missed_by_speed_metrics</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>small_n_24;single_session;self_reported_understanding</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>participant_ai_familiarity</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1_study_design=1 · D2_sample_quality=2 · D3_measurement_validity=2 · D4_temporal_strength=1 · D5_directness=1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=partial_includes_cs_students · subject_match=programming_adjacent · tool_match=copilot_like_generative_ai · scope=risk_identification</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Contradicted</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Despite faster first-task completion, participants struggled to understand and debug AI-generated solutions and reported low ownership of the final program - documenting metacognitive and dependence risks that pure speed metrics miss.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3491101.3519665</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3491101.3519665">https://dl.acm.org/doi/10.1145/3491101.3519665</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3491101.3519665" title="Expectation vs. Experience: Evaluating the Usability of Code Generation Tools"><code>S-2022-vaithilingam</code></a></td></tr></tbody></table></div></div></section><section id="full-03-quality-en" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>03 Evidence Quality, Counterevidence &amp; Method Audit</h2><p class="full-chapter-lead">Examine why evidence is credible, where it conflicts, and which conclusions require downgrading.</p></header><div class="full-chapter-body"><section class="method-review-group"><header class="method-review-title"><h3>Audit target: overall</h3><span class="method-verdict">Concern</span></header><div class="method-audit-grid"><article class="method-audit-item method-met"><div class="method-audit-head"><strong>Control Group</strong><span class="method-status">Met</span></div><p>All three studies include a no-AI control group.</p></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>Randomization</strong><span class="method-status">Met</span></div><p>Kazemitabaar 2023 and Bastani 2025 use randomized assignment.</p></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>Pre Test</strong><span class="method-status">Met</span></div><p>Kazemitabaar 2023 has a pre-study evaluation; Bastani 2025 measures baseline covariates.</p></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>Post Test</strong><span class="method-status">Met</span></div><p>Immediate post-tests present in all studies.</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Retention Test</strong><span class="method-status">Partial</span></div><p>Kazemitabaar 2023 has 1-week retention; Bastani 2025 has no delayed test; Marzuki 2024 has delayed test.</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Transfer Test</strong><span class="method-status">Partial</span></div><p>Kazemitabaar 2023 code-modification task is transfer-adjacent; no full no-AI transfer task.</p></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>Sample Bias</strong><span class="method-status">Met</span></div><p>Bastani 2025 nearly 1000 students; Kazemitabaar 2023 small (69) young sample.</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Self Selection</strong><span class="method-status">Partial</span></div><p>Marzuki 2024 consent-based participation risks self-selection.</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Measurement Validity</strong><span class="method-status">Partial</span></div><p>Practice/task performance is not equated to learning; independent exams present in Bastani 2025 only.</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Confounders</strong><span class="method-status">Partial</span></div><p>Prior programming competency interacts with AI benefit in Kazemitabaar 2023.</p></article><article class="method-audit-item method-not_applicable"><div class="method-audit-head"><strong>Instructor Effect</strong><span class="method-status">N/A</span></div><p>Kazemitabaar 2023 is self-paced; classroom studies may carry instructor effects.</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Novelty Effect</strong><span class="method-status">Partial</span></div><p>Short interventions likely inflate engagement; none of the studies controlled for novelty.</p></article><article class="method-audit-item method-not_applicable"><div class="method-audit-head"><strong>Tool Version Effect</strong><span class="method-status">N/A</span></div><p>Single tool versions studied; rapid tool change limits durability.</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Ai Usage Policy</strong><span class="method-status">Partial</span></div><p>Bastani 2025 explicitly contrasts unguarded vs guardrailed usage policies.</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Dropout</strong><span class="method-status">Partial</span></div><p>Marzuki 2024 reports attrition; others not detailed.</p></article></div><div class="method-guard-wrap"><strong>Task vs learning guard: </strong><p class="method-guard">Bastani 2025 demonstrates the danger of equating the two: +48-127% practice performance coexisted with -17% independent exam performance.</p></div></section><article class="conflict-card"><strong>Tribunal note: </strong><div class="conflict-text"><p>Disagreement comes from outcome separation (task vs learning), tool design (guarded vs unguarded), and population (K-12 / professionals vs university). Task-performance evidence is consistently positive across randomized and benchmark studies;…</p><details class="detail-expander"><summary>Expand full explanation</summary><div class="detail-body"><p>Disagreement comes from outcome separation (task vs learning), tool design (guarded vs unguarded), and population (K-12 / professionals vs university). Task-performance evidence is consistently positive across randomized and benchmark studies; the only study measuring independent performance after AI removal shows harm without guardrails; usability and artifact studies add dependence and quality caveats rather than resolving the learning question.</p></div></details></div></article><div class="evidence-tribunal" data-visual="evidence-tribunal-grid"><p class="tribunal-summary"><strong>Decision: </strong>Pilot · <strong>Confidence: </strong>Moderate</p><div class="tribunal-grid"><article class="tribunal-card supported"><header><span aria-hidden="true">✓</span><h3>Can claim</h3><span class="tribunal-count">7</span></header><ul><li><p>AI coding assistants reliably increase task performance during training (completion speed, correctness) — E-001, E-006.</p><div class="tribunal-evidence-refs"><code>E-001</code><code>E-006</code></div></li><li><p>Unguarded generative AI access can harm independent problem solving when access is removed — E-004.</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li><li><p>Guardrail design (hints instead of answers) substantially mitigates the negative learning effect — E-005.</p><div class="tribunal-evidence-refs"><code>E-005</code></div></li><li><p>Task performance gains do not automatically imply learning gains — E-004 vs E-006 (within-study contrast).</p><div class="tribunal-evidence-refs"><code>E-004</code><code>E-006</code></div></li><li><p>Tool capability is substantial: Codex solves roughly half to three-quarters of CS1 exam-style questions — E-010.</p><div class="tribunal-evidence-refs"><code>E-010</code></div></li><li><p>Professional-developer RCT shows ~55% faster task completion with Copilot; directness limited by professional population — E-008.</p><div class="tribunal-evidence-refs"><code>E-008</code></div></li><li><p>LLM code explanations rate comparable to student-authored explanations, viable as scaffold material — E-011.</p><div class="tribunal-evidence-refs"><code>E-011</code></div></li></ul></article><article class="tribunal-card uncertain"><header><span aria-hidden="true">?</span><h3>Cannot yet claim</h3><span class="tribunal-count">4</span></header><ul><li><p>Whether AI coding assistants improve or preserve actual programming learning in university novices — no direct university-level RCT in reviewed set [无直接证据]</p></li><li><p>Whether one-week neutral retention (Kazemitabaar 2023) extends to a semester — E-003.</p><div class="tribunal-evidence-refs"><code>E-003</code></div></li><li><p>Whether benchmark quality findings (E-009) and explanation-quality ratings (E-011) translate into classroom learning gains.</p><div class="tribunal-evidence-refs"><code>E-009</code><code>E-011</code></div></li><li><p>How comprehension/ownership difficulties documented in usability studies (E-012) behave over a full semester with guardrails.</p><div class="tribunal-evidence-refs"><code>E-012</code></div></li></ul></article><article class="tribunal-card contradicted"><header><span aria-hidden="true">×</span><h3>Contradicted claims</h3><span class="tribunal-count">2</span></header><ul><li><p>The claim &#x27;AI tools always improve learning&#x27; is contradicted by E-004 (unguarded access, -17% independent exam).</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li><li><p>The claim &#x27;speed gains equal learning gains&#x27; is contradicted by the task-vs-learning separation across E-001/E-006/E-008 vs E-004.</p><div class="tribunal-evidence-refs"><code>E-001</code><code>E-006</code><code>E-008</code><code>E-004</code></div></li></ul></article><article class="tribunal-card missing"><header><span aria-hidden="true">…</span><h3>Missing evidence</h3><span class="tribunal-count">4</span></header><ul><li><p>RCT of AI coding assistants in university programming courses with retention and no-AI transfer tests.</p></li><li><p>Studies varying AI usage policy within the same course.</p></li><li><p>Longitudinal data on AI dependency beyond one course.</p></li><li><p>Peer-reviewed replication of the professional speed RCT (Peng et al. remains a preprint).</p></li></ul></article></div><details class="supporting-visual"><summary>EvidenceFlow Protocol</summary><svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="EvidenceFlow Protocol"><title>EvidenceFlow Protocol</title><desc>Research flow from framing, retrieval, fetch/verify, extraction, challenge, method audit and adjudication to applicability and intervention evaluation.</desc><defs><marker id="arr-en-full-workflow" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">EvidenceFlow Protocol</text><rect x="24" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="56.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="56.0" y="126.0">Frame</tspan></text><line x1="88" y1="126" x2="100" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="100" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="132.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="132.0" y="126.0">Retrieve</tspan></text><line x1="164" y1="126" x2="176" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="176" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="208.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="208.0" y="126.0">Fetch</tspan><tspan x="208.0" y="140.0">Verify</tspan></text><line x1="240" y1="126" x2="252" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="252" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="284.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="284.0" y="126.0">Extract</tspan></text><line x1="316" y1="126" x2="328" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="328" y="100" width="64" height="52" rx="8" fill="#5E8A6A"/><text x="360.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="360.0" y="126.0">Challenge</tspan></text><line x1="392" y1="126" x2="404" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="404" y="100" width="64" height="52" rx="8" fill="#5E8A6A"/><text x="436.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="436.0" y="126.0">Audit</tspan><tspan x="436.0" y="140.0">Method</tspan></text><line x1="468" y1="126" x2="480" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="480" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="512.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="512.0" y="126.0">Adjudicate</tspan></text><line x1="544" y1="126" x2="556" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="556" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="588.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="588.0" y="126.0">Applicability</tspan></text><line x1="620" y1="126" x2="632" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="632" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="664.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="664.0" y="126.0">Intervene</tspan><tspan x="664.0" y="140.0">Evaluate</tspan></text></svg></details><details class="supporting-visual"><summary>Tribunal infographic</summary><svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="Evidence Tribunal infographic"><title>Evidence Tribunal infographic</title><desc>Evidence IDs for claims that can and cannot be claimed, plus the recommended action badge; full claim text is in the tribunal cards below.</desc><defs><marker id="arr-en-full-tribunal" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">Evidence Tribunal</text><text x="24" y="180" font-size="13" font-weight="700" fill="#3A3833">Source of conflict</text><text x="24" y="202" font-size="11" fill="#8A867E">See tribunal cards below</text><rect x="24" y="216" width="180" height="28" rx="14" fill="#C99A4A"/><text x="114" y="235" text-anchor="middle" font-size="13" font-weight="700" fill="#fff">PILOT</text><text x="250" y="80" font-size="13" font-weight="700" fill="#5E8A6A">Can claim (7)</text><circle cx="254" cy="100" r="3" fill="#5E8A6A"/><text x="266" y="105" font-size="11" fill="#3A3833">E-001</text><circle cx="254" cy="122" r="3" fill="#5E8A6A"/><text x="266" y="127" font-size="11" fill="#3A3833">E-006</text><circle cx="254" cy="144" r="3" fill="#5E8A6A"/><text x="266" y="149" font-size="11" fill="#3A3833">E-004</text><circle cx="254" cy="166" r="3" fill="#5E8A6A"/><text x="266" y="171" font-size="11" fill="#3A3833">E-005</text><text x="490" y="80" font-size="13" font-weight="700" fill="#A85B53">Cannot claim (2)</text><circle cx="494" cy="100" r="3" fill="#A85B53"/><text x="506" y="105" font-size="11" fill="#3A3833">E-004</text><circle cx="494" cy="122" r="3" fill="#A85B53"/><text x="506" y="127" font-size="11" fill="#3A3833">E-001</text><circle cx="494" cy="144" r="3" fill="#A85B53"/><text x="506" y="149" font-size="11" fill="#3A3833">E-006</text><circle cx="494" cy="166" r="3" fill="#A85B53"/><text x="506" y="171" font-size="11" fill="#3A3833">E-008</text></svg></details></div><div class="trace-chain" data-visual="trace-chain"><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-001</strong><p class="trace-claim-text">AI coding assistants significantly increase task completion speed and completion rate during training.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-001</code><span>Completion time</span><span class="dir pos">Positive effect</span><span class="trace-relation">claim relation: Support</span><span>Q 9.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3544548.3580919"><code>S-2023-kazemitabaar</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-002</strong><p class="trace-claim-text">Access to AI code generation did not decrease performance on manual code-modification tasks.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-002</code><span>Independent problem solving</span><span class="dir neu">Null effect</span><span class="trace-relation">claim relation: Neutral</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3544548.3580919"><code>S-2023-kazemitabaar</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-003</strong><p class="trace-claim-text">One week after training, retention differences between Codex and baseline groups did not reach statistical significance.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-003</code><span>Retention</span><span class="dir neu">Null effect</span><span class="trace-relation">claim relation: Neutral</span><span>Q 9.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3544548.3580919"><code>S-2023-kazemitabaar</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-004</strong><p class="trace-claim-text">Students with unguarded GPT-4 access performed 17% worse on the independent exam than the control group, despite higher practice performance.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-004</code><span>Independent problem solving</span><span class="dir neg">Negative effect</span><span class="trace-relation">claim relation: Contradict</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122"><code>S-2025-bastani</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-005</strong><p class="trace-claim-text">Guardrail design of the AI tutor (hints instead of answers, teacher-informed prompts) largely eliminated the negative learning effect.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-005</code><span>Independent problem solving</span><span class="dir neu">Null effect</span><span class="trace-relation">claim relation: Support</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122"><code>S-2025-bastani</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-006</strong><p class="trace-claim-text">Access to GPT-4 during practice improves task performance (48% for GPT Base, 127% for GPT Tutor) — but this task performance does not transfer to independent exam performance.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-006</code><span>Assignment score</span><span class="dir pos">Positive effect</span><span class="trace-relation">claim relation: Support</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122"><code>S-2025-bastani</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-007</strong><p class="trace-claim-text">ChatGPT as a formative feedback tool produced a significant positive impact on students&#x27; academic writing skills with positive student perceptions.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-007</code><span>Knowledge gain</span><span class="dir pos">Positive effect</span><span class="trace-relation">claim relation: Support</span><span>Q 6.0</span><span class="trace-arrow">→</span><a href="https://link.springer.com/article/10.1186/s40561-024-00295-9"><code>S-2024-marzuki</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-008</strong><p class="trace-claim-text">Professional developers with Copilot access completed a standardized coding task about 55% faster than the control group (RCT, n=95).</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-008</code><span>Completion time</span><span class="dir pos">Positive effect</span><span class="trace-relation">claim relation: Support</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://doi.org/10.48550/arXiv.2302.06590"><code>S-2023-peng</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-009</strong><p class="trace-claim-text">Systematic benchmark evaluation reports mixed quality results for Copilot-generated code relative to human code: correctness competitive on parts of the benchmark while security-relevant defects are documented.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-009</code><span>Code quality</span><span class="dir neu">Null effect</span><span class="trace-relation">claim relation: Neutral</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://doi.org/10.1016/j.jss.2023.111734"><code>S-2023-yetistiren</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-010</strong><p class="trace-claim-text">Codex produced passing-level solutions for roughly half to three-quarters of CS1 exam-style questions depending on the dataset, indicating substantial task-capability headroom available to novices.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-010</code><span>Assignment score</span><span class="dir pos">Positive effect</span><span class="trace-relation">claim relation: Support</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3545945.3569830"><code>S-2022-finnie-ansley</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-011</strong><div class="trace-claim-text"><p>Controlled comparisons find LLM-generated code explanations comparable to (in places better than) student-authored explanations, suggesting viability as explanatory scaffold material rather than as a replacement for student explanation prac…</p><details class="detail-expander"><summary>Expand full explanation</summary><div class="detail-body"><p>Controlled comparisons find LLM-generated code explanations comparable to (in places better than) student-authored explanations, suggesting viability as explanatory scaffold material rather than as a replacement for student explanation practice.</p></div></details></div></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-011</code><span>Metacognition</span><span class="dir pos">Positive effect</span><span class="trace-relation">claim relation: Support</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3587102.3588785"><code>S-2023-explanations-compare</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-012</strong><p class="trace-claim-text">Despite faster first-task completion, participants struggled to understand and debug AI-generated solutions and reported low ownership of the final program - documenting metacognitive and dependence risks that pure speed metrics miss.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-012</code><span>Over-reliance</span><span class="dir neg">Negative effect</span><span class="trace-relation">claim relation: Contradict</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3491101.3519665"><code>S-2022-vaithilingam</code></a></div></div></article></div><div id="chart-trace-en" class="chart-mount" aria-label="Claim-Evidence Trace"></div><p class="chart-interpretation"><strong>What this means: </strong>Every important claim must resolve to Evidence IDs and original sources.</p></div></section><section id="full-04-action-en" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>04 Applicability &amp; Teaching Action</h2><p class="full-chapter-lead">Connect applicability, guardrails and teaching actions to specific evidence.</p></header><div class="full-chapter-body"><p><strong>Target population: </strong>university first-year computer science students learning C programming for the first time</p>
2509
+ </div><div class="scope-grid"><article class="scope-card"><h3>Research question</h3><p class="scope-text">我们准备在大学一年级 C 语言课程中允许学生使用生成式 AI 编程助手。它到底会不会提高学习效果?应该怎样引入?</p></article><article class="scope-card"><h3>Target learners</h3><p class="scope-text">Education level: First-year undergraduate; Major: Computer science; Prior knowledge: First programming course, no prior text-based programming; Learner characteristics: Mixed-ability large class (60 students)</p></article><article class="scope-card"><h3>Course context</h3><p class="scope-text">Subject: C programming; Course type: Lecture + lab; Duration: 16 weeks (one semester)</p></article><article class="scope-card"><h3>AI intervention</h3><p class="scope-text">Method: Lecture with lab exercises; AI tool: Generative AI coding assistant; Allowed usage: Under design (pending evidence review); Frequency: Weekly lab sessions; Duration: One semester</p></article><article class="scope-card"><h3>Comparison</h3><p class="scope-text">No AI Coding Assistant Control</p></article><article class="scope-card"><h3>Outcome constructs</h3><p class="scope-text">Primary outcomes: Independent problem solving, Code quality; Secondary outcomes: Completion time, Retention, Knowledge gain; Risk outcomes: AI dependency, Over-reliance, Reduced transfer</p></article><article class="scope-card"><h3>Research scope</h3><p class="scope-text">Time range: 2021 2026; Geography: Worldwide; Study designs: Randomized controlled trial, Quasi-experimental, Observational</p></article><article class="scope-card"><h3>Decision success condition</h3><p class="scope-text">independent problem solving and code quality improve (or do not decline) while AI dependency risk stays controlled; evidence base supports a bounded pilot.</p></article></div></div></section><section id="full-02-evidence-en" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>02 Key Evidence &amp; Outcome Separation</h2><p class="full-chapter-lead">Place task performance, actual learning, retention and risk on one evidence map without conflating them.</p></header><div class="full-chapter-body"><div class="retrieval-grid"><article><h3>Inclusion criteria</h3><ul><li>Studies Of Generative AI Coding Tools In Learning To Program</li><li>Outcomes Measuring Learning Not Only Task Speed</li><li>University Or Novice Programming Populations</li></ul></article><article><h3>Exclusion criteria</h3><ul><li>Practitioner Anecdotes Without Data</li><li>Industry Professional Populations Only</li></ul></article></div><div class="retrieval-coverage"><h3>Source coverage</h3><p><code>S-2023-kazemitabaar</code> <code>S-2025-bastani</code> <code>S-2024-marzuki</code> <code>S-2023-peng</code> <code>S-2023-yetistiren</code> <code>S-2022-finnie-ansley</code> <code>S-2023-explanations-compare</code> <code>S-2022-vaithilingam</code></p><p class="retrieval-note">This report shows only retrieval metadata present in result; it does not fabricate PRISMA/funnel counts when none exist.</p></div><div class="outcome-separation" data-visual="outcome-separation"><div class="visual-heading"><h3>Outcome Separation · Task performance ≠ learning</h3><p>Outcomes are adjudicated separately so faster training performance is not silently treated as evidence of learning.</p><p class="semantic-note">Effect direction comes from evidence.effect_direction; supporting a claim does not imply a positive outcome.</p></div><div class="outcome-groups"><article class="outcome-group outcome-task"><h3>Task / proximal performance</h3><ul><li><strong>Completion time</strong><span class="outcome-states"><span class="dir pos">Positive effect 2</span></span></li><li><strong>Code quality</strong><span class="outcome-states"><span class="dir neu">Null effect 1</span></span></li><li><strong>Assignment score</strong><span class="outcome-states"><span class="dir pos">Positive effect 2</span></span></li></ul></article><article class="outcome-group outcome-learning"><h3>Learning / retention / transfer</h3><ul><li><strong>Knowledge gain</strong><span class="outcome-states"><span class="dir pos">Positive effect 1</span></span></li><li><strong>Retention</strong><span class="outcome-states"><span class="dir neu">Null effect 1</span></span></li><li><strong>Independent problem solving</strong><span class="outcome-states"><span class="dir neg">Negative effect 1</span><span class="dir neu">Null effect 2</span></span></li></ul></article><article class="outcome-group outcome-risk"><h3>Risk / dependency</h3><ul><li><strong>Over-reliance</strong><span class="outcome-states"><span class="dir neg">Negative effect 1</span></span></li></ul></article><article class="outcome-group outcome-other"><h3>Other outcomes</h3><ul><li><strong>Metacognition</strong><span class="outcome-states"><span class="dir pos">Positive effect 1</span></span></li></ul></article></div></div><div class='table-wrap outcome-table'><table class='data-table'><thead><tr><th>Outcome</th><th>Positive effect</th><th>Negative effect</th><th>Null effect</th><th>Evidence</th></tr></thead><tbody><tr><td><strong>Knowledge gain</strong><span class='raw-tag' title='raw id'>knowledge_gain</span></td><td class='num'>1</td><td class='num'>0</td><td class='num'>0</td><td><code>E-007</code> </td></tr><tr><td><strong>Retention</strong><span class='raw-tag' title='raw id'>retention</span></td><td class='num'>0</td><td class='num'>0</td><td class='num'>1</td><td><code>E-003</code> </td></tr><tr><td><strong>Independent problem solving</strong><span class='raw-tag' title='raw id'>independent_problem_solving</span></td><td class='num'>0</td><td class='num'>1</td><td class='num'>2</td><td><code>E-002</code> <code>E-004</code> <code>E-005</code> </td></tr><tr><td><strong>Completion time</strong><span class='raw-tag' title='raw id'>completion_time</span></td><td class='num'>2</td><td class='num'>0</td><td class='num'>0</td><td><code>E-001</code> <code>E-008</code> </td></tr><tr><td><strong>Code quality</strong><span class='raw-tag' title='raw id'>code_quality</span></td><td class='num'>0</td><td class='num'>0</td><td class='num'>1</td><td><code>E-009</code> </td></tr><tr><td><strong>Assignment score</strong><span class='raw-tag' title='raw id'>assignment_score</span></td><td class='num'>2</td><td class='num'>0</td><td class='num'>0</td><td><code>E-006</code> <code>E-010</code> </td></tr><tr><td><strong>Metacognition</strong><span class='raw-tag' title='raw id'>metacognition</span></td><td class='num'>1</td><td class='num'>0</td><td class='num'>0</td><td><code>E-011</code> </td></tr><tr><td><strong>Over-reliance</strong><span class='raw-tag' title='raw id'>over_reliance</span></td><td class='num'>0</td><td class='num'>1</td><td class='num'>0</td><td><code>E-012</code> </td></tr></tbody></table></div><div class="visual-surface" data-visual="outcome-evidence-balance"><svg viewBox="0 0 720 300" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="Outcome evidence effect balance"><title>Outcome evidence effect balance</title><desc>Positive / negative / null effect counts per outcome (based on effect_direction, not claim support).</desc><rect x="0" y="0" width="720" height="300" fill="#FFFFFF"/><line x1="439.0" y1="46" x2="439.0" y2="182" stroke="#999" stroke-width="1" stroke-dasharray="3,3"/><text x="190" y="58.0" text-anchor="end" font-size="11" fill="#333">Knowledge gain</text><rect x="439.0" y="49.8" width="120.5" height="9.4" fill="#5E8A6A"/><text x="499.2" y="58.0" text-anchor="middle" font-size="9" fill="#fff">1</text><text x="190" y="75.0" text-anchor="end" font-size="11" fill="#333">Retention</text><text x="190" y="92.0" text-anchor="end" font-size="11" fill="#333">Independent problem solving</text><rect x="318.5" y="83.8" width="120.5" height="9.4" fill="#A85B53"/><text x="378.8" y="92.0" text-anchor="middle" font-size="9" fill="#fff">-1</text><text x="190" y="109.0" text-anchor="end" font-size="11" fill="#333">Completion time</text><rect x="439.0" y="100.8" width="241.0" height="9.4" fill="#5E8A6A"/><text x="559.5" y="109.0" text-anchor="middle" font-size="9" fill="#fff">2</text><text x="190" y="126.0" text-anchor="end" font-size="11" fill="#333">Code quality</text><text x="190" y="143.0" text-anchor="end" font-size="11" fill="#333">Assignment score</text><rect x="439.0" y="134.8" width="241.0" height="9.4" fill="#5E8A6A"/><text x="559.5" y="143.0" text-anchor="middle" font-size="9" fill="#fff">2</text><text x="190" y="160.0" text-anchor="end" font-size="11" fill="#333">Metacognition</text><rect x="439.0" y="151.8" width="120.5" height="9.4" fill="#5E8A6A"/><text x="499.2" y="160.0" text-anchor="middle" font-size="9" fill="#fff">1</text><text x="190" y="177.0" text-anchor="end" font-size="11" fill="#333">Over-reliance</text><rect x="318.5" y="168.8" width="120.5" height="9.4" fill="#A85B53"/><text x="378.8" y="177.0" text-anchor="middle" font-size="9" fill="#fff">-1</text><rect x="439.0" y="206.1" width="8.0" height="6" fill="#C99A4A"/><rect x="439.0" y="214.9" width="8.0" height="6" fill="#C99A4A"/><rect x="439.0" y="232.4" width="8.0" height="6" fill="#C99A4A"/><rect x="198" y="14" width="10" height="10" fill="#5E8A6A"/><text x="212" y="23" font-size="10" fill="#333">Positive effect</text><rect x="395" y="14" width="10" height="10" fill="#A85B53"/><text x="409" y="23" font-size="10" fill="#333">Negative effect</text><rect x="592" y="14" width="10" height="10" fill="#C99A4A"/><text x="606" y="23" font-size="10" fill="#333">Null effect</text></svg><p class="chart-interpretation"><strong>What this means: </strong>Positive / negative / null effect-direction evidence counts per outcome. This visual encodes effect_direction, not whether evidence supports a claim.</p></div><div id="chart-outcome-en" class="chart-mount" aria-label="Outcome Evidence Overview"></div><figure class="academic-figure" data-visual="outcome-evidence-balance"><svg viewBox="0 0 1518 300" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="Effect direction by outcome type (publication figure)"><title>Effect direction by outcome type (publication figure)</title><desc>Counts of positive / negative / null effects per outcome type with an integer count axis, theme-independent. Source: EduEvidence result.json.</desc><rect width="1518" height="300" fill="#FFFFFF"/><line x1="70" y1="250" x2="650" y2="250" stroke="#333" stroke-width="1"/><text x="106.2" y="266" text-anchor="middle" font-size="10" fill="#333">Knowledge gain</text><rect x="88.1" y="150.0" width="18.1" height="100.0" fill="#38BDF8"/><text x="97.2" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><rect x="106.2" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="124.4" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><text x="178.8" y="266" text-anchor="middle" font-size="10" fill="#333">Retention</text><rect x="160.6" y="250.0" width="18.1" height="1.0" fill="#38BDF8"/><rect x="178.8" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="196.9" y="150.0" width="18.1" height="100.0" fill="#F59E0B"/><text x="205.9" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><text x="251.2" y="266" text-anchor="middle" font-size="10" fill="#333">Independent problem solving</text><rect x="233.1" y="250.0" width="18.1" height="1.0" fill="#38BDF8"/><rect x="251.2" y="150.0" width="18.1" height="100.0" fill="#10B981"/><text x="260.3" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><rect x="269.4" y="50.0" width="18.1" height="200.0" fill="#F59E0B"/><text x="278.4" y="47.0" text-anchor="middle" font-size="9" fill="#333">2</text><text x="323.8" y="266" text-anchor="middle" font-size="10" fill="#333">Completion time</text><rect x="305.6" y="50.0" width="18.1" height="200.0" fill="#38BDF8"/><text x="314.7" y="47.0" text-anchor="middle" font-size="9" fill="#333">2</text><rect x="323.8" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="341.9" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><text x="396.2" y="266" text-anchor="middle" font-size="10" fill="#333">Code quality</text><rect x="378.1" y="250.0" width="18.1" height="1.0" fill="#38BDF8"/><rect x="396.2" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="414.4" y="150.0" width="18.1" height="100.0" fill="#F59E0B"/><text x="423.4" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><text x="468.8" y="266" text-anchor="middle" font-size="10" fill="#333">Assignment score</text><rect x="450.6" y="50.0" width="18.1" height="200.0" fill="#38BDF8"/><text x="459.7" y="47.0" text-anchor="middle" font-size="9" fill="#333">2</text><rect x="468.8" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="486.9" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><text x="541.2" y="266" text-anchor="middle" font-size="10" fill="#333">Metacognition</text><rect x="523.1" y="150.0" width="18.1" height="100.0" fill="#38BDF8"/><text x="532.2" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><rect x="541.2" y="250.0" width="18.1" height="1.0" fill="#10B981"/><rect x="559.4" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><text x="613.8" y="266" text-anchor="middle" font-size="10" fill="#333">Over-reliance</text><rect x="595.6" y="250.0" width="18.1" height="1.0" fill="#38BDF8"/><rect x="613.8" y="150.0" width="18.1" height="100.0" fill="#10B981"/><text x="622.8" y="147.0" text-anchor="middle" font-size="9" fill="#333">1</text><rect x="631.9" y="250.0" width="18.1" height="1.0" fill="#F59E0B"/><line x1="65" y1="250.0" x2="70" y2="250.0" stroke="#999"/><text x="62" y="253.0" text-anchor="end" font-size="9" fill="#666">0</text><line x1="65" y1="150.0" x2="70" y2="150.0" stroke="#999"/><text x="62" y="153.0" text-anchor="end" font-size="9" fill="#666">1</text><line x1="65" y1="50.0" x2="70" y2="50.0" stroke="#999"/><text x="62" y="53.0" text-anchor="end" font-size="9" fill="#666">2</text><text x="360.0" y="30" text-anchor="middle" font-size="14" font-weight="700" fill="#111">Effect direction by outcome type</text><rect x="70" y="8" width="10" height="10" fill="#38BDF8"/><text x="84" y="17" font-size="10" fill="#333">Positive effect</text><rect x="267" y="8" width="10" height="10" fill="#10B981"/><text x="281" y="17" font-size="10" fill="#333">Negative effect</text><rect x="464" y="8" width="10" height="10" fill="#F59E0B"/><text x="478" y="17" font-size="10" fill="#333">Null effect</text><text x="20" y="290" font-size="11" fill="#333333" font-style="italic">Fig. 1. Counts of positive / negative / null effects per outcome type (based on effect_direction; publication figure, theme-independent). Source: EduEvidence result.json.</text></svg><figcaption>Fig. 1. Positive / negative / null effect counts per outcome (based on effect_direction, not claim support).</figcaption></figure><div class='matrix-controls'><div class='matrix-tools'><input id='matrix-search-full-en' type='search' placeholder='Search evidence…' aria-label='Filter / search evidence'><select id='matrix-direction-full-en' aria-label='Filter by effect direction'><option value=''>All effects</option><option value='positive'>Positive effect</option><option value='negative'>Negative effect</option><option value='null'>Null effect</option></select><select id='matrix-outcome-full-en' aria-label='Filter by outcome type'><option value=''>All outcomes</option><option value='assignment_score'>Assignment score</option><option value='code_quality'>Code quality</option><option value='completion_time'>Completion time</option><option value='independent_problem_solving'>Independent problem solving</option><option value='knowledge_gain'>Knowledge gain</option><option value='metacognition'>Metacognition</option><option value='over_reliance'>Over-reliance</option><option value='retention'>Retention</option></select></div></div><div class='table-wrap matrix-wrap'><table id='evidence-matrix-full-en' class='data-table evidence-matrix'><thead><tr><th>ID</th><th>Outcome</th><th>Effect</th><th>Quality</th><th>Claim</th><th>Source</th></tr></thead><tbody><tr data-effect="positive" data-direction="positive" data-outcome="completion_time" data-search="e-001 study-kazemitabaar-2023 smpl-kazemitabaar-2023-n69 studying the effect of ai code generators on supporting novice learners in introductory programming ai coding assistants significantly increase task completion speed and completion rate during training. 69 novices ages 10-17 with no prior text-based programming experience access_to_openai_codex_ai_coding_assistant_during_training baseline_group_without_ai_coding_assistant positive s-2023-kazemitabaar"><td><code>E-001</code></td><td><strong>Completion time</strong></td><td><span class="dir pos">Positive effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">9</span><span class="quality-meter" aria-hidden="true"><i style="width:90%"></i></span></div></td><td class="claim-cell"><p>AI coding assistants significantly increase task completion speed and completion rate during training.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-KAZEMITABAAR-2023</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-KAZEMITABAAR-2023-N69</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Randomized controlled trial</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>K-12 Ages 10 17</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>69 novices ages 10-17 with no prior text-based programming experience</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>69</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>Access To Openai Codex AI Coding Assistant During Training</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>Baseline Group Without AI Coding Assistant</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>Code Authoring Task Progress And Time</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>1.15x completion rate, 0.57x time, 1.8x correctness</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Positive effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>3 Weeks Training</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>controlled experiment with random assignment, immediate post-test and 1-week retention test</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>randomized_controlled_design;immediate_post_test_and_retention_test;code_modification_task_guard</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>non_university_population_ages_10_17;small_sample_69;self-paced environment differs from classroom</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>Prior Programming Competency Interaction</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 2 · D2 Sample quality = 2 · D3 Measurement validity = 2 · D4 Temporal strength = 2 · D5 Directness = 1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>9.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=partial_novice_programmers_but_younger · subject_match=introductory_programming · tool_match=codex_like_generative_ai · scope=task_performance_during_training</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.7</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>AI coding assistants significantly increase task completion speed and completion rate during training.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3544548.3580919</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">https://dl.acm.org/doi/10.1145/3544548.3580919</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3544548.3580919" title="Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming"><code>S-2023-kazemitabaar</code></a></td></tr><tr data-effect="null" data-direction="null" data-outcome="independent_problem_solving" data-search="e-002 study-kazemitabaar-2023 smpl-kazemitabaar-2023-n69 studying the effect of ai code generators on supporting novice learners in introductory programming access to ai code generation did not decrease performance on manual code-modification tasks. 69 novices ages 10-17 with no prior text-based programming experience access_to_openai_codex_ai_coding_assistant_during_training baseline_group_without_ai_coding_assistant null s-2023-kazemitabaar"><td><code>E-002</code></td><td><strong>Independent problem solving</strong></td><td><span class="dir neu">Null effect</span><span class="relation-note">Claim relation: Neutral</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>Access to AI code generation did not decrease performance on manual code-modification tasks.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-KAZEMITABAAR-2023</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-KAZEMITABAAR-2023-N69</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Randomized controlled trial</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>K-12 Ages 10 17</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>69 novices ages 10-17 with no prior text-based programming experience</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>69</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>Access To Openai Codex AI Coding Assistant During Training</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>Baseline Group Without AI Coding Assistant</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>manual code-modification tasks during training</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>no significant difference between groups</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Null effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Neutral</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>3 Weeks Training</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>controlled experiment, code-modification task followed each code-authoring task</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>direct_test_of_transfer-adjacent_skill;same_session_measurement</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>code modification is not full independent problem solving;non_university population</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>Practice Effect</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 2 · D2 Sample quality = 2 · D3 Measurement validity = 1 · D4 Temporal strength = 1 · D5 Directness = 1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=partial · subject_match=introductory_programming · tool_match=codex · scope=short-term manual code modification</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.5</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Access to AI code generation did not decrease performance on manual code-modification tasks.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3544548.3580919</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">https://dl.acm.org/doi/10.1145/3544548.3580919</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3544548.3580919" title="Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming"><code>S-2023-kazemitabaar</code></a></td></tr><tr data-effect="null" data-direction="null" data-outcome="retention" data-search="e-003 study-kazemitabaar-2023 smpl-kazemitabaar-2023-n69 studying the effect of ai code generators on supporting novice learners in introductory programming one week after training, retention differences between codex and baseline groups did not reach statistical significance. 69 novices ages 10-17 with no prior text-based programming experience access_to_openai_codex_ai_coding_assistant_during_training baseline_group_without_ai_coding_assistant null s-2023-kazemitabaar"><td><code>E-003</code></td><td><strong>Retention</strong></td><td><span class="dir neu">Null effect</span><span class="relation-note">Claim relation: Neutral</span></td><td><div class="quality-cell"><span class="num">9</span><span class="quality-meter" aria-hidden="true"><i style="width:90%"></i></span></div></td><td class="claim-cell"><p>One week after training, retention differences between Codex and baseline groups did not reach statistical significance.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-KAZEMITABAAR-2023</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-KAZEMITABAAR-2023-N69</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Randomized controlled trial</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>K-12 Ages 10 17</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>69 novices ages 10-17 with no prior text-based programming experience</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>69</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>Access To Openai Codex AI Coding Assistant During Training</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>Baseline Group Without AI Coding Assistant</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>retention post-test one week after training</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>slightly better for Codex group but not significant</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Null effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Neutral</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>3 Weeks Training Plus 1 Week Retention</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>controlled experiment with delayed retention test</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>delayed_test_included</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>1-week retention window is short;small sample;non-university population</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>Prior Competency</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 2 · D2 Sample quality = 2 · D3 Measurement validity = 2 · D4 Temporal strength = 2 · D5 Directness = 1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>9.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=partial · subject_match=introductory_programming · tool_match=codex · scope=retention over one week</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.5</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>One week after training, retention differences between Codex and baseline groups did not reach statistical significance.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3544548.3580919</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">https://dl.acm.org/doi/10.1145/3544548.3580919</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3544548.3580919" title="Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming"><code>S-2023-kazemitabaar</code></a></td></tr><tr data-effect="negative" data-direction="negative" data-outcome="independent_problem_solving" data-search="e-004 study-bastani-2025 smpl-bastani-2025-n950 generative ai without guardrails can harm learning: evidence from high school mathematics students with unguarded gpt-4 access performed 17% worse on the independent exam than the control group, despite higher practice performance. nearly a thousand high school math students in turkey gpt4_based_tutor_gpt_base_unguarded no_generative_ai_control negative s-2025-bastani"><td><code>E-004</code></td><td><strong>Independent problem solving</strong></td><td><span class="dir neg">Negative effect</span><span class="relation-note">Claim relation: Contradict</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>Students with unguarded GPT-4 access performed 17% worse on the independent exam than the control group, despite higher practice performance.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-BASTANI-2025</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-BASTANI-2025-N950</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Randomized controlled trial</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>High school</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>nearly a thousand high school math students in Turkey</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>950</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>Gpt4 Based Tutor Gpt Base Unguarded</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>No Generative AI Control</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>exam without access to AI resources after practice phase</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>negative_17_percent_on_independent_exam</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Negative effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Contradict</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>In Class Study Sessions</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>large-scale randomized controlled trial, practice phase then closed-book exam</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>large_scale_rct;independent_exam_without_ai;arm_wise_design</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>high_school_mathematics_not_university_programming;single_country</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>Tool Design Difference</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 2 · D2 Sample quality = 2 · D3 Measurement validity = 2 · D4 Temporal strength = 1 · D5 Directness = 1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=partial_same_age_band_different_subject · subject_match=no_mathematics_vs_programming · tool_match=gpt4_chat_interface · scope=unguarded_general_chat_interface</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.75</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Students with unguarded GPT-4 access performed 17% worse on the independent exam than the control group, despite higher practice performance.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://www.pnas.org/doi/10.1073/pnas.2422633122</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">https://www.pnas.org/doi/10.1073/pnas.2422633122</a></dd></div></dl></details></td><td><a class="source-link" href="https://www.pnas.org/doi/10.1073/pnas.2422633122" title="Generative AI without guardrails can harm learning: Evidence from high school mathematics"><code>S-2025-bastani</code></a></td></tr><tr data-effect="null" data-direction="null" data-outcome="independent_problem_solving" data-search="e-005 study-bastani-2025 smpl-bastani-2025-n950 generative ai without guardrails can harm learning: evidence from high school mathematics guardrail design of the ai tutor (hints instead of answers, teacher-informed prompts) largely eliminated the negative learning effect. nearly a thousand high school math students in turkey gpt4_tutor_with_teacher_designed_guardrails no_generative_ai_control null s-2025-bastani"><td><code>E-005</code></td><td><strong>Independent problem solving</strong></td><td><span class="dir neu">Null effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>Guardrail design of the AI tutor (hints instead of answers, teacher-informed prompts) largely eliminated the negative learning effect.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-BASTANI-2025</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-BASTANI-2025-N950</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Randomized controlled trial</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>High school</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>nearly a thousand high school math students in Turkey</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>950</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>Gpt4 Tutor With Teacher Designed Guardrails</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>No Generative AI Control</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>exam without access to AI resources after practice phase</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>negative effect essentially eradicated, no positive effect observed</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Null effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>In Class Study Sessions</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>large-scale randomized controlled trial, three arms</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>direct_manipulation_of_tool_design;large_sample</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>no_positive_learning_gain_even_with_guardrails;subject_mismatch</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>Prompt Engineering Effort</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 2 · D2 Sample quality = 2 · D3 Measurement validity = 2 · D4 Temporal strength = 1 · D5 Directness = 1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=partial · subject_match=no · tool_match=guardrailed_tutor_design · scope=guardrail_design_principle_transferable</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.75</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Guardrail design of the AI tutor (hints instead of answers, teacher-informed prompts) largely eliminated the negative learning effect.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://www.pnas.org/doi/10.1073/pnas.2422633122</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">https://www.pnas.org/doi/10.1073/pnas.2422633122</a></dd></div></dl></details></td><td><a class="source-link" href="https://www.pnas.org/doi/10.1073/pnas.2422633122" title="Generative AI without guardrails can harm learning: Evidence from high school mathematics"><code>S-2025-bastani</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="assignment_score" data-search="e-006 study-bastani-2025 smpl-bastani-2025-n950 generative ai without guardrails can harm learning: evidence from high school mathematics access to gpt-4 during practice improves task performance (48% for gpt base, 127% for gpt tutor) — but this task performance does not transfer to independent exam performance. nearly a thousand high school math students in turkey gpt4_tutor_access_during_practice no_generative_ai_control positive s-2025-bastani"><td><code>E-006</code></td><td><strong>Assignment score</strong></td><td><span class="dir pos">Positive effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>Access to GPT-4 during practice improves task performance (48% for GPT Base, 127% for GPT Tutor) — but this task performance does not transfer to independent exam performance.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-BASTANI-2025</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-BASTANI-2025-N950</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Randomized controlled trial</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>High school</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>nearly a thousand high school math students in Turkey</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>950</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>Gpt4 Tutor Access During Practice</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>No Generative AI Control</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>practice problem performance during study sessions</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>48-127 percent improvement on practice problems</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Positive effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>In Class Study Sessions</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>randomized controlled trial with practice and closed-book exam phases</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>same_study_compares_task_and_learning;large_sample</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>subject_mismatch_mathematics</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>Task Familiarity</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 2 · D2 Sample quality = 2 · D3 Measurement validity = 2 · D4 Temporal strength = 1 · D5 Directness = 1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>strong</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=partial · subject_match=no · tool_match=gpt4 · scope=task_performance_vs_learning_separation</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.75</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Access to GPT-4 during practice improves task performance (48% for GPT Base, 127% for GPT Tutor) — but this task performance does not transfer to independent exam performance.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://www.pnas.org/doi/10.1073/pnas.2422633122</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">https://www.pnas.org/doi/10.1073/pnas.2422633122</a></dd></div></dl></details></td><td><a class="source-link" href="https://www.pnas.org/doi/10.1073/pnas.2422633122" title="Generative AI without guardrails can harm learning: Evidence from high school mathematics"><code>S-2025-bastani</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="knowledge_gain" data-search="e-007 study-marzuki-2024 smpl-marzuki-2024-n72 impact of chatgpt on esl students&#x27; academic writing skills chatgpt as a formative feedback tool produced a significant positive impact on students&#x27; academic writing skills with positive student perceptions. undergraduate esl students at an indian university chatgpt_as_formative_feedback_tool traditional_instruction_control positive s-2024-marzuki"><td><code>E-007</code></td><td><strong>Knowledge gain</strong></td><td><span class="dir pos">Positive effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">6</span><span class="quality-meter" aria-hidden="true"><i style="width:60%"></i></span></div></td><td class="claim-cell"><p>ChatGPT as a formative feedback tool produced a significant positive impact on students&#x27; academic writing skills with positive student perceptions.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-MARZUKI-2024</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-MARZUKI-2024-N72</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Impact of ChatGPT on ESL students&#x27; academic writing skills</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Impact of ChatGPT on ESL students&#x27; academic writing skills</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2024</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Mixed methods</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>Undergraduate</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>undergraduate ESL students at an Indian university</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>72</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>Chatgpt As Formative Feedback Tool</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>Traditional Instruction Control</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>writing tests with pre-post-delayed design</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>significant positive impact on writing skills</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Positive effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>6 Hours Intervention</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>mixed methods intervention study, pre/post/delayed tests and focus groups</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>delayed_post_test;mixed_methods_triangulation</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>short_intervention_6_hours;single_institution;elite_private_university</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>Self Selection Consent</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 1 · D2 Sample quality = 1 · D3 Measurement validity = 2 · D4 Temporal strength = 2 · D5 Directness = 0</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>6.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=yes_undergraduate · subject_match=no_writing_not_programming · tool_match=chatgpt · scope=formative_feedback_writing</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>ChatGPT as a formative feedback tool produced a significant positive impact on students&#x27; academic writing skills with positive student perceptions.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://link.springer.com/article/10.1186/s40561-024-00295-9</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://link.springer.com/article/10.1186/s40561-024-00295-9">https://link.springer.com/article/10.1186/s40561-024-00295-9</a></dd></div></dl></details></td><td><a class="source-link" href="https://link.springer.com/article/10.1186/s40561-024-00295-9" title="Impact of ChatGPT on ESL students&#x27; academic writing skills"><code>S-2024-marzuki</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="completion_time" data-search="e-008 study-peng-2023 smpl-peng-2023-n95 the impact of ai on developer productivity: evidence from github copilot professional developers with copilot access completed a standardized coding task about 55% faster than the control group (rct, n=95). 95 recruited professional developers completing a standardized coding task on a freelance platform access_to_github_copilot_during_task control_group_without_copilot positive s-2023-peng"><td><code>E-008</code></td><td><strong>Completion time</strong></td><td><span class="dir pos">Positive effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>Professional developers with Copilot access completed a standardized coding task about 55% faster than the control group (RCT, n=95).</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-PENG-2023</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-PENG-2023-N95</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>The Impact of AI on Developer Productivity: Evidence from GitHub Copilot</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>The Impact of AI on Developer Productivity: Evidence from GitHub Copilot</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Randomized controlled trial</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>Professional Developers Not Students</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>95 recruited professional developers completing a standardized coding task on a freelance platform</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>95</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>Access To Github Copilot During Task</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>Control Group Without Copilot</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>Time To Complete Http Server Implementation</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>~55.8% faster task completion in Copilot group</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Positive effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>Single Task Session</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>online randomized controlled experiment with objective completion-time metric</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>randomized_controlled_design;objective_completion_time_metric</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>professional_population_not_students;single_task_ecology;preprint_not_peer_reviewed</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>Task Familiarity;Platform Recruitment Self Selection</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 2 · D2 Sample quality = 2 · D3 Measurement validity = 2 · D4 Temporal strength = 1 · D5 Directness = 1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=mismatch_professional_developers · subject_match=adjacent_web_development_task · tool_match=copilot_like_generative_ai · scope=task_performance_only_no_learning_outcome</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.6</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Professional developers with Copilot access completed a standardized coding task about 55% faster than the control group (RCT, n=95).</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://doi.org/10.48550/arXiv.2302.06590</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://doi.org/10.48550/arXiv.2302.06590">https://doi.org/10.48550/arXiv.2302.06590</a></dd></div></dl></details></td><td><a class="source-link" href="https://doi.org/10.48550/arXiv.2302.06590" title="The Impact of AI on Developer Productivity: Evidence from GitHub Copilot"><code>S-2023-peng</code></a></td></tr><tr data-effect="null" data-direction="null" data-outcome="code_quality" data-search="e-009 study-yetistiren-2023 smpl-yetistiren-2023-bench github copilot ai pair programmer: asset or liability? systematic benchmark evaluation reports mixed quality results for copilot-generated code relative to human code: correctness competitive on parts of the benchmark while security-relevant defects are documented. copilot-generated and human-written programs drawn from published benchmark datasets copilot_generated_programs human_written_programs_on_same_benchmarks null s-2023-yetistiren"><td><code>E-009</code></td><td><strong>Code quality</strong></td><td><span class="dir neu">Null effect</span><span class="relation-note">Claim relation: Neutral</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>Systematic benchmark evaluation reports mixed quality results for Copilot-generated code relative to human code: correctness competitive on parts of the benchmark while security-relevant defects are documented.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-YETISTIREN-2023</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-YETISTIREN-2023-BENCH</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>GitHub Copilot AI Pair Programmer: Asset or Liability?</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>GitHub Copilot AI Pair Programmer: Asset or Liability?</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Observational</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>Not Applicable Code Artifacts</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>Copilot-generated and human-written programs drawn from published benchmark datasets</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>Copilot Generated Programs</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>Human Written Programs On Same Benchmarks</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>Correctness Security Maintainability Metrics On Benchmarks</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>mixed quality profile; no single-direction summary</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Null effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Neutral</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>Not Applicable Artifact Study</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>systematic empirical evaluation of generated code against human baselines on public benchmarks</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>multi_dimensional_quality_metrics;reproducible_benchmark_protocol</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>artifact_benchmark_not_classroom;no_learning_outcome;tool_version_from_2023</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>Benchmark Task Distribution</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 1 · D2 Sample quality = 2 · D3 Measurement validity = 2 · D4 Temporal strength = 1 · D5 Directness = 1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=mismatch_no_learners_in_study · subject_match=introductory_adjacent_code_tasks · tool_match=copilot_like_generative_ai · scope=output_quality_only</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Systematic benchmark evaluation reports mixed quality results for Copilot-generated code relative to human code: correctness competitive on parts of the benchmark while security-relevant defects are documented.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://doi.org/10.1016/j.jss.2023.111734</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://doi.org/10.1016/j.jss.2023.111734">https://doi.org/10.1016/j.jss.2023.111734</a></dd></div></dl></details></td><td><a class="source-link" href="https://doi.org/10.1016/j.jss.2023.111734" title="GitHub Copilot AI Pair Programmer: Asset or Liability?"><code>S-2023-yetistiren</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="assignment_score" data-search="e-010 study-finnieansley-2022 smpl-finnieansley-2022-qsets using github copilot to solve introductory programming problems codex produced passing-level solutions for roughly half to three-quarters of cs1 exam-style questions depending on the dataset, indicating substantial task-capability headroom available to novices. cs1 exam-style question sets answered by codex and compared against published student score distributions codex_answer_generation_on_cs1_questions published_student_cohort_score_distributions positive s-2022-finnie-ansley"><td><code>E-010</code></td><td><strong>Assignment score</strong></td><td><span class="dir pos">Positive effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>Codex produced passing-level solutions for roughly half to three-quarters of CS1 exam-style questions depending on the dataset, indicating substantial task-capability headroom available to novices.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-FINNIEANSLEY-2022</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-FINNIEANSLEY-2022-QSETS</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Using GitHub Copilot to Solve Introductory Programming Problems</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Using GitHub Copilot to Solve Introductory Programming Problems</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2022</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Observational</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>University Year 1 Question Sets</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>CS1 exam-style question sets answered by Codex and compared against published student score distributions</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>Codex Answer Generation On CS1 Questions</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>Published Student Cohort Score Distributions</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>Pass Rate On CS1 Exam Style Questions</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>passing solutions on ~50-75% of questions across datasets</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Positive effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>Not Applicable Capability Probe</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>capability benchmark against published student distributions; reproducible question sets</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>public_reproducible_question_sets;directly_relevant_task_domain</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>tool_solves_task_does_not_equate_student_learning;codex_2021_model_version_outdated</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>Question Leakage Into Training Data Possible</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 1 · D2 Sample quality = 2 · D3 Measurement validity = 2 · D4 Temporal strength = 1 · D5 Directness = 1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=partial_measures_tool_not_students · subject_match=introductory_programming · tool_match=copilot_like_generative_ai · scope=tool_capability_headroom</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Codex produced passing-level solutions for roughly half to three-quarters of CS1 exam-style questions depending on the dataset, indicating substantial task-capability headroom available to novices.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3545945.3569830</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3545945.3569830">https://dl.acm.org/doi/10.1145/3545945.3569830</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3545945.3569830" title="Using GitHub Copilot to Solve Introductory Programming Problems"><code>S-2022-finnie-ansley</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="metacognition" data-search="e-011 study-explcomp-2023 smpl-explcomp-2023-ratings comparing code explanations created by students and large language models controlled comparisons find llm-generated code explanations comparable to (in places better than) student-authored explanations, suggesting viability as explanatory scaffold material rather than as a replacement for student explanation practice. student-produced versus llm-produced explanations of short programs under controlled comparison llm_generated_code_explanations student_generated_explanations_of_same_programs positive s-2023-explanations-compare"><td><code>E-011</code></td><td><strong>Metacognition</strong></td><td><span class="dir pos">Positive effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>Controlled comparisons find LLM-generated code explanations comparable to (in places better than) student-authored explanations, suggesting viability as explanatory scaffold material rather than as a replacemen…</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-EXPLCOMP-2023</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-EXPLCOMP-2023-RATINGS</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Comparing Code Explanations Created by Students and Large Language Models</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Comparing Code Explanations Created by Students and Large Language Models</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Observational</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>University Introductory</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>Student-produced versus LLM-produced explanations of short programs under controlled comparison</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>LLM Generated Code Explanations</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>Student Generated Explanations Of Same Programs</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>Rated Explanation Quality And Comprehensibility</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>comparable-or-better rated quality vs student explanations</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Positive effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>Single Session Ratings</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>controlled comparison with blind rating of explanation pairs</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>controlled_pairwise_comparison;learning_process_relevant_construct</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>short_term_ratings_not_learning_gains;small_program_snippets_ecology</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>Rating Criteria Subjectivity</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 1 · D2 Sample quality = 2 · D3 Measurement validity = 2 · D4 Temporal strength = 1 · D5 Directness = 1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=partial_scaffold_material_only · subject_match=introductory_programming · tool_match=llm_explanations · scope=scaffold_quality_not_effectiveness</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Controlled comparisons find LLM-generated code explanations comparable to (in places better than) student-authored explanations, suggesting viability as explanatory scaffold material rather than as a replacement for student explanation practice.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3587102.3588785</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3587102.3588785">https://dl.acm.org/doi/10.1145/3587102.3588785</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3587102.3588785" title="Comparing Code Explanations Created by Students and Large Language Models"><code>S-2023-explanations-compare</code></a></td></tr><tr data-effect="negative" data-direction="negative" data-outcome="over_reliance" data-search="e-012 study-vaithilingam-2022 smpl-vaithilingam-2022-n24 expectation vs. experience: evaluating the usability of code generation tools despite faster first-task completion, participants struggled to understand and debug ai-generated solutions and reported low ownership of the final program - documenting metacognitive and dependence risks that pure speed metrics miss. 24 participants in a within-subjects usability study of copilot-style tools copilot_assisted_program_writing within_subject_baseline_without_tool negative s-2022-vaithilingam"><td><code>E-012</code></td><td><strong>Over-reliance</strong></td><td><span class="dir neg">Negative effect</span><span class="relation-note">Claim relation: Contradict</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>Despite faster first-task completion, participants struggled to understand and debug AI-generated solutions and reported low ownership of the final program - documenting metacognitive and dependence risks that…</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>STUDY-VAITHILINGAM-2022</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SMPL-VAITHILINGAM-2022-N24</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Expectation vs. Experience: Evaluating the Usability of Code Generation Tools</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Expectation vs. Experience: Evaluating the Usability of Code Generation Tools</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2022</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Qualitative</dd></div><div class="evidence-detail-row"><dt>Education level</dt><dd>Mixed Cs Students And Professionals</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>24 participants in a within-subjects usability study of Copilot-style tools</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>24</dd></div><div class="evidence-detail-row"><dt>Intervention</dt><dd>Copilot Assisted Program Writing</dd></div><div class="evidence-detail-row"><dt>Comparison</dt><dd>Within Subject Baseline Without Tool</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>Understanding Ownership And Debugging Reports</dd></div><div class="evidence-detail-row"><dt>Effect / result</dt><dd>documented comprehension/ownership difficulties despite speed gain</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Negative effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Contradict</dd></div><div class="evidence-detail-row"><dt>Duration</dt><dd>Single Session</dd></div><div class="evidence-detail-row"><dt>Method</dt><dd>within-subject usability study with tasks, observation and interviews</dd></div><div class="evidence-detail-row"><dt>Strengths</dt><dd>rich_qualitative_process_data;constructs_missed_by_speed_metrics</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>small_n_24;single_session;self_reported_understanding</dd></div><div class="evidence-detail-row"><dt>Confounders</dt><dd>Participant AI Familiarity</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 1 · D2 Sample quality = 2 · D3 Measurement validity = 2 · D4 Temporal strength = 1 · D5 Directness = 1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>Evidence level</dt><dd>moderate</dd></div><div class="evidence-detail-row"><dt>Applicability</dt><dd>learner_match=partial_includes_cs_students · subject_match=programming_adjacent · tool_match=copilot_like_generative_ai · scope=risk_identification</dd></div><div class="evidence-detail-row"><dt>Confidence</dt><dd>0.55</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Contradicted</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Despite faster first-task completion, participants struggled to understand and debug AI-generated solutions and reported low ownership of the final program - documenting metacognitive and dependence risks that pure speed metrics miss.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3491101.3519665</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3491101.3519665">https://dl.acm.org/doi/10.1145/3491101.3519665</a></dd></div></dl></details></td><td><a class="source-link" href="https://dl.acm.org/doi/10.1145/3491101.3519665" title="Expectation vs. Experience: Evaluating the Usability of Code Generation Tools"><code>S-2022-vaithilingam</code></a></td></tr></tbody></table></div></div></section><section id="full-03-quality-en" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>03 Evidence Quality, Counterevidence &amp; Method Audit</h2><p class="full-chapter-lead">Examine why evidence is credible, where it conflicts, and which conclusions require downgrading.</p></header><div class="full-chapter-body"><section class="method-review-group"><header class="method-review-title"><h3>Audit target: overall</h3><span class="method-verdict">Concern</span></header><div class="method-audit-grid"><article class="method-audit-item method-met"><div class="method-audit-head"><strong>Control Group</strong><span class="method-status">Met</span></div><p>All three studies include a no-AI control group.</p></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>Randomization</strong><span class="method-status">Met</span></div><p>Kazemitabaar 2023 and Bastani 2025 use randomized assignment.</p></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>Pre Test</strong><span class="method-status">Met</span></div><p>Kazemitabaar 2023 has a pre-study evaluation; Bastani 2025 measures baseline covariates.</p></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>Post Test</strong><span class="method-status">Met</span></div><p>Immediate post-tests present in all studies.</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Retention Test</strong><span class="method-status">Partial</span></div><p>Kazemitabaar 2023 has 1-week retention; Bastani 2025 has no delayed test; Marzuki 2024 has delayed test.</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Transfer Test</strong><span class="method-status">Partial</span></div><p>Kazemitabaar 2023 code-modification task is transfer-adjacent; no full no-AI transfer task.</p></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>Sample Bias</strong><span class="method-status">Met</span></div><p>Bastani 2025 nearly 1000 students; Kazemitabaar 2023 small (69) young sample.</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Self Selection</strong><span class="method-status">Partial</span></div><p>Marzuki 2024 consent-based participation risks self-selection.</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Measurement Validity</strong><span class="method-status">Partial</span></div><p>Practice/task performance is not equated to learning; independent exams present in Bastani 2025 only.</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Confounders</strong><span class="method-status">Partial</span></div><p>Prior programming competency interacts with AI benefit in Kazemitabaar 2023.</p></article><article class="method-audit-item method-not_applicable"><div class="method-audit-head"><strong>Instructor Effect</strong><span class="method-status">N/A</span></div><p>Kazemitabaar 2023 is self-paced; classroom studies may carry instructor effects.</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Novelty Effect</strong><span class="method-status">Partial</span></div><p>Short interventions likely inflate engagement; none of the studies controlled for novelty.</p></article><article class="method-audit-item method-not_applicable"><div class="method-audit-head"><strong>Tool Version Effect</strong><span class="method-status">N/A</span></div><p>Single tool versions studied; rapid tool change limits durability.</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Ai Usage Policy</strong><span class="method-status">Partial</span></div><p>Bastani 2025 explicitly contrasts unguarded vs guardrailed usage policies.</p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Dropout</strong><span class="method-status">Partial</span></div><p>Marzuki 2024 reports attrition; others not detailed.</p></article></div><div class="method-guard-wrap"><strong>Task vs learning guard: </strong><p class="method-guard">Bastani 2025 demonstrates the danger of equating the two: +48-127% practice performance coexisted with -17% independent exam performance.</p></div></section><article class="conflict-card"><strong>Tribunal note: </strong><div class="conflict-text"><p>Disagreement comes from outcome separation (task vs learning), tool design (guarded vs unguarded), and population (K-12 / professionals vs university). Task-performance evidence is consistently positive across randomized and benchmark studies;…</p><details class="detail-expander"><summary>Expand full explanation</summary><div class="detail-body"><p>Disagreement comes from outcome separation (task vs learning), tool design (guarded vs unguarded), and population (K-12 / professionals vs university). Task-performance evidence is consistently positive across randomized and benchmark studies; the only study measuring independent performance after AI removal shows harm without guardrails; usability and artifact studies add dependence and quality caveats rather than resolving the learning question.</p></div></details></div></article><div class="evidence-tribunal" data-visual="evidence-tribunal-grid"><p class="tribunal-summary"><strong>Decision: </strong>Pilot · <strong>Confidence: </strong>Moderate</p><div class="tribunal-grid"><article class="tribunal-card supported"><header><span aria-hidden="true">✓</span><h3>Can claim</h3><span class="tribunal-count">7</span></header><ul><li><p>AI coding assistants reliably increase task performance during training (completion speed, correctness) — E-001, E-006.</p><div class="tribunal-evidence-refs"><code>E-001</code><code>E-006</code></div></li><li><p>Unguarded generative AI access can harm independent problem solving when access is removed — E-004.</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li><li><p>Guardrail design (hints instead of answers) substantially mitigates the negative learning effect — E-005.</p><div class="tribunal-evidence-refs"><code>E-005</code></div></li><li><p>Task performance gains do not automatically imply learning gains — E-004 vs E-006 (within-study contrast).</p><div class="tribunal-evidence-refs"><code>E-004</code><code>E-006</code></div></li><li><p>Tool capability is substantial: Codex solves roughly half to three-quarters of CS1 exam-style questions — E-010.</p><div class="tribunal-evidence-refs"><code>E-010</code></div></li><li><p>Professional-developer RCT shows ~55% faster task completion with Copilot; directness limited by professional population — E-008.</p><div class="tribunal-evidence-refs"><code>E-008</code></div></li><li><p>LLM code explanations rate comparable to student-authored explanations, viable as scaffold material — E-011.</p><div class="tribunal-evidence-refs"><code>E-011</code></div></li></ul></article><article class="tribunal-card uncertain"><header><span aria-hidden="true">?</span><h3>Cannot yet claim</h3><span class="tribunal-count">4</span></header><ul><li><p>Whether AI coding assistants improve or preserve actual programming learning in university novices — no direct university-level RCT in reviewed set [无直接证据]</p></li><li><p>Whether one-week neutral retention (Kazemitabaar 2023) extends to a semester — E-003.</p><div class="tribunal-evidence-refs"><code>E-003</code></div></li><li><p>Whether benchmark quality findings (E-009) and explanation-quality ratings (E-011) translate into classroom learning gains.</p><div class="tribunal-evidence-refs"><code>E-009</code><code>E-011</code></div></li><li><p>How comprehension/ownership difficulties documented in usability studies (E-012) behave over a full semester with guardrails.</p><div class="tribunal-evidence-refs"><code>E-012</code></div></li></ul></article><article class="tribunal-card contradicted"><header><span aria-hidden="true">×</span><h3>Contradicted claims</h3><span class="tribunal-count">2</span></header><ul><li><p>The claim &#x27;AI tools always improve learning&#x27; is contradicted by E-004 (unguarded access, -17% independent exam).</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li><li><p>The claim &#x27;speed gains equal learning gains&#x27; is contradicted by the task-vs-learning separation across E-001/E-006/E-008 vs E-004.</p><div class="tribunal-evidence-refs"><code>E-001</code><code>E-006</code><code>E-008</code><code>E-004</code></div></li></ul></article><article class="tribunal-card missing"><header><span aria-hidden="true">…</span><h3>Missing evidence</h3><span class="tribunal-count">4</span></header><ul><li><p>RCT of AI coding assistants in university programming courses with retention and no-AI transfer tests.</p></li><li><p>Studies varying AI usage policy within the same course.</p></li><li><p>Longitudinal data on AI dependency beyond one course.</p></li><li><p>Peer-reviewed replication of the professional speed RCT (Peng et al. remains a preprint).</p></li></ul></article></div><details class="supporting-visual"><summary>EvidenceFlow Protocol</summary><svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="EvidenceFlow Protocol"><title>EvidenceFlow Protocol</title><desc>Research flow from framing, retrieval, fetch/verify, extraction, challenge, method audit and adjudication to applicability and intervention evaluation.</desc><defs><marker id="arr-en-full-workflow" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">EvidenceFlow Protocol</text><rect x="24" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="56.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="56.0" y="126.0">Frame</tspan></text><line x1="88" y1="126" x2="100" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="100" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="132.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="132.0" y="126.0">Retrieve</tspan></text><line x1="164" y1="126" x2="176" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="176" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="208.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="208.0" y="126.0">Fetch</tspan><tspan x="208.0" y="140.0">Verify</tspan></text><line x1="240" y1="126" x2="252" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="252" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="284.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="284.0" y="126.0">Extract</tspan></text><line x1="316" y1="126" x2="328" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="328" y="100" width="64" height="52" rx="8" fill="#5E8A6A"/><text x="360.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="360.0" y="126.0">Challenge</tspan></text><line x1="392" y1="126" x2="404" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="404" y="100" width="64" height="52" rx="8" fill="#5E8A6A"/><text x="436.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="436.0" y="126.0">Audit</tspan><tspan x="436.0" y="140.0">Method</tspan></text><line x1="468" y1="126" x2="480" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="480" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="512.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="512.0" y="126.0">Adjudicate</tspan></text><line x1="544" y1="126" x2="556" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="556" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="588.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="588.0" y="126.0">Applicability</tspan></text><line x1="620" y1="126" x2="632" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="632" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="664.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="664.0" y="126.0">Intervene</tspan><tspan x="664.0" y="140.0">Evaluate</tspan></text></svg></details><details class="supporting-visual"><summary>Tribunal infographic</summary><svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="Evidence Tribunal infographic"><title>Evidence Tribunal infographic</title><desc>Evidence IDs for claims that can and cannot be claimed, plus the recommended action badge; full claim text is in the tribunal cards below.</desc><defs><marker id="arr-en-full-tribunal" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">Evidence Tribunal</text><text x="24" y="180" font-size="13" font-weight="700" fill="#3A3833">Source of conflict</text><text x="24" y="202" font-size="11" fill="#8A867E">See tribunal cards below</text><rect x="24" y="216" width="180" height="28" rx="14" fill="#C99A4A"/><text x="114" y="235" text-anchor="middle" font-size="13" font-weight="700" fill="#fff">Pilot</text><text x="250" y="80" font-size="13" font-weight="700" fill="#5E8A6A">Can claim (7)</text><circle cx="254" cy="100" r="3" fill="#5E8A6A"/><text x="266" y="105" font-size="11" fill="#3A3833">E-001</text><circle cx="254" cy="122" r="3" fill="#5E8A6A"/><text x="266" y="127" font-size="11" fill="#3A3833">E-006</text><circle cx="254" cy="144" r="3" fill="#5E8A6A"/><text x="266" y="149" font-size="11" fill="#3A3833">E-004</text><circle cx="254" cy="166" r="3" fill="#5E8A6A"/><text x="266" y="171" font-size="11" fill="#3A3833">E-005</text><text x="490" y="80" font-size="13" font-weight="700" fill="#A85B53">Cannot claim (2)</text><circle cx="494" cy="100" r="3" fill="#A85B53"/><text x="506" y="105" font-size="11" fill="#3A3833">E-004</text><circle cx="494" cy="122" r="3" fill="#A85B53"/><text x="506" y="127" font-size="11" fill="#3A3833">E-001</text><circle cx="494" cy="144" r="3" fill="#A85B53"/><text x="506" y="149" font-size="11" fill="#3A3833">E-006</text><circle cx="494" cy="166" r="3" fill="#A85B53"/><text x="506" y="171" font-size="11" fill="#3A3833">E-008</text></svg></details></div><div class="trace-chain" data-visual="trace-chain"><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-001</strong><p class="trace-claim-text">AI coding assistants significantly increase task completion speed and completion rate during training.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-001</code><span>Completion time</span><span class="dir pos">Positive effect</span><span class="trace-relation">claim relation: Support</span><span>Q 9.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3544548.3580919"><code>S-2023-kazemitabaar</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-002</strong><p class="trace-claim-text">Access to AI code generation did not decrease performance on manual code-modification tasks.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-002</code><span>Independent problem solving</span><span class="dir neu">Null effect</span><span class="trace-relation">claim relation: Neutral</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3544548.3580919"><code>S-2023-kazemitabaar</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-003</strong><p class="trace-claim-text">One week after training, retention differences between Codex and baseline groups did not reach statistical significance.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-003</code><span>Retention</span><span class="dir neu">Null effect</span><span class="trace-relation">claim relation: Neutral</span><span>Q 9.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3544548.3580919"><code>S-2023-kazemitabaar</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-004</strong><p class="trace-claim-text">Students with unguarded GPT-4 access performed 17% worse on the independent exam than the control group, despite higher practice performance.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-004</code><span>Independent problem solving</span><span class="dir neg">Negative effect</span><span class="trace-relation">claim relation: Contradict</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122"><code>S-2025-bastani</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-005</strong><p class="trace-claim-text">Guardrail design of the AI tutor (hints instead of answers, teacher-informed prompts) largely eliminated the negative learning effect.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-005</code><span>Independent problem solving</span><span class="dir neu">Null effect</span><span class="trace-relation">claim relation: Support</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122"><code>S-2025-bastani</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-006</strong><p class="trace-claim-text">Access to GPT-4 during practice improves task performance (48% for GPT Base, 127% for GPT Tutor) — but this task performance does not transfer to independent exam performance.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-006</code><span>Assignment score</span><span class="dir pos">Positive effect</span><span class="trace-relation">claim relation: Support</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122"><code>S-2025-bastani</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-007</strong><p class="trace-claim-text">ChatGPT as a formative feedback tool produced a significant positive impact on students&#x27; academic writing skills with positive student perceptions.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-007</code><span>Knowledge gain</span><span class="dir pos">Positive effect</span><span class="trace-relation">claim relation: Support</span><span>Q 6.0</span><span class="trace-arrow">→</span><a href="https://link.springer.com/article/10.1186/s40561-024-00295-9"><code>S-2024-marzuki</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-008</strong><p class="trace-claim-text">Professional developers with Copilot access completed a standardized coding task about 55% faster than the control group (RCT, n=95).</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-008</code><span>Completion time</span><span class="dir pos">Positive effect</span><span class="trace-relation">claim relation: Support</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://doi.org/10.48550/arXiv.2302.06590"><code>S-2023-peng</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-009</strong><p class="trace-claim-text">Systematic benchmark evaluation reports mixed quality results for Copilot-generated code relative to human code: correctness competitive on parts of the benchmark while security-relevant defects are documented.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-009</code><span>Code quality</span><span class="dir neu">Null effect</span><span class="trace-relation">claim relation: Neutral</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://doi.org/10.1016/j.jss.2023.111734"><code>S-2023-yetistiren</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-010</strong><p class="trace-claim-text">Codex produced passing-level solutions for roughly half to three-quarters of CS1 exam-style questions depending on the dataset, indicating substantial task-capability headroom available to novices.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-010</code><span>Assignment score</span><span class="dir pos">Positive effect</span><span class="trace-relation">claim relation: Support</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3545945.3569830"><code>S-2022-finnie-ansley</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-011</strong><div class="trace-claim-text"><p>Controlled comparisons find LLM-generated code explanations comparable to (in places better than) student-authored explanations, suggesting viability as explanatory scaffold material rather than as a replacement for student explanation prac…</p><details class="detail-expander"><summary>Expand full explanation</summary><div class="detail-body"><p>Controlled comparisons find LLM-generated code explanations comparable to (in places better than) student-authored explanations, suggesting viability as explanatory scaffold material rather than as a replacement for student explanation practice.</p></div></details></div></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-011</code><span>Metacognition</span><span class="dir pos">Positive effect</span><span class="trace-relation">claim relation: Support</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3587102.3588785"><code>S-2023-explanations-compare</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-012</strong><p class="trace-claim-text">Despite faster first-task completion, participants struggled to understand and debug AI-generated solutions and reported low ownership of the final program - documenting metacognitive and dependence risks that pure speed metrics miss.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-012</code><span>Over-reliance</span><span class="dir neg">Negative effect</span><span class="trace-relation">claim relation: Contradict</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://dl.acm.org/doi/10.1145/3491101.3519665"><code>S-2022-vaithilingam</code></a></div></div></article></div><div id="chart-trace-en" class="chart-mount" aria-label="Claim-Evidence Trace"></div><p class="chart-interpretation"><strong>What this means: </strong>Every important claim must resolve to Evidence IDs and original sources.</p></div></section><section id="full-04-action-en" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>04 Applicability &amp; Teaching Action</h2><p class="full-chapter-lead">Connect applicability, guardrails and teaching actions to specific evidence.</p></header><div class="full-chapter-body"><p><strong>Target population: </strong>university first-year computer science students learning C programming for the first time</p>
2190
2510
  <p><strong>Target context: </strong>16-week lecture-lab course, 60 students, TA support, offline</p>
2191
2511
  <p><strong>Suitable for: </strong>pilot in first-year C course with guardrailed usage policy</p>
2192
2512
  <p><strong>Not suitable for: </strong>unrestricted AI adoption without usage policy</p>
@@ -2221,7 +2541,7 @@ select:focus-visible,
2221
2541
  <p><strong>Success threshold: </strong>treatment group shows non-inferior independent problem solving (delta within 5%) AND superior or equal retention AND AI dependency index below threshold; if independent problem solving declines &gt;10%, the pilot is judged unsuccessful regardless of task-performance gains.</p>
2222
2542
  <p><strong>Analysis plan: </strong>pre-registered comparison of treatment vs comparison sections on baseline-adjusted learning metrics (ANCOVA); task-performance metrics reported separately from learning metrics; subgroup analysis by prior programming competency; stop-condition monitoring at weeks 3, 5, 7.</p>
2223
2543
  <h3>Evaluation design infographic</h3>
2224
- <svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="Evaluation design flow"><title>Evaluation design flow</title><desc>Evaluation flow across baseline, post test, retention and transfer; full metrics and analysis plan are in the evaluation section.</desc><defs><marker id="arr-en-full-evaluation" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">Evaluation Design Flow</text><rect x="30" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="105.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="105.0" y="138.0">Baseline</tspan></text><text x="105.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">pre-test</text><line x1="180" y1="138" x2="192" y2="138" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-evaluation)"/><rect x="192" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="267.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="267.0" y="138.0">Post test</tspan></text><text x="267.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">post-test</text><line x1="342" y1="138" x2="354" y2="138" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-evaluation)"/><rect x="354" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="429.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="429.0" y="138.0">Retention</tspan></text><text x="429.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">retention test</text><line x1="504" y1="138" x2="516" y2="138" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-evaluation)"/><rect x="516" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="591.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="591.0" y="138.0">Transfer</tspan></text><text x="591.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">transfer test (no AI)</text></svg><div class="visual-suppressed"><strong>Benchmark visual suppressed</strong><p>result.json carries no benchmark.baselines, so this visual is omitted; see the standalone benchmark report.</p></div></div></section><section id="full-06-sources-en" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>06 Sources, Traceability &amp; Appendix</h2><p class="full-chapter-lead">Preserve original sources, URLs, evidence IDs and retrieval metadata for auditability.</p></header><div class="full-chapter-body"><h3>Source list</h3><div class='table-wrap'><table class='data-table source-table'><thead><tr><th>ID</th><th>Title</th><th>Year</th><th>Authority</th><th>Verifiable location</th></tr></thead><tbody><tr><td><code>S-2023-kazemitabaar</code></td><td class='cell-main source-title-cell'>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>View source &amp; provenance</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>Original title</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="source-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="source-detail-row"><dt>Authority</dt><dd>Tier 1 DOI-verified paper</dd></div><div class="source-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3544548.3580919</dd></div><div class="source-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">https://dl.acm.org/doi/10.1145/3544548.3580919</a></dd></div></dl></details></td><td>2023</td><td>Tier 1 DOI-verified paper</td><td class='cell-main'><a href='https://dl.acm.org/doi/10.1145/3544548.3580919'>https://dl.acm.org/doi/10.1145/3544548.3580919</a></td></tr><tr><td><code>S-2025-bastani</code></td><td class='cell-main source-title-cell'>Generative AI without guardrails can harm learning: Evidence from high school mathematics <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>View source &amp; provenance</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>Original title</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="source-detail-row"><dt>Year</dt><dd>2025</dd></div><div class="source-detail-row"><dt>Authority</dt><dd>Tier 1 DOI-verified paper</dd></div><div class="source-detail-row"><dt>Source location</dt><dd>https://www.pnas.org/doi/10.1073/pnas.2422633122</dd></div><div class="source-detail-row"><dt>Verifiable link</dt><dd><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">https://www.pnas.org/doi/10.1073/pnas.2422633122</a></dd></div></dl></details></td><td>2025</td><td>Tier 1 DOI-verified paper</td><td class='cell-main'><a href='https://www.pnas.org/doi/10.1073/pnas.2422633122'>https://www.pnas.org/doi/10.1073/pnas.2422633122</a></td></tr><tr><td><code>S-2024-marzuki</code></td><td class='cell-main source-title-cell'>Impact of ChatGPT on ESL students&#x27; academic writing skills <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>View source &amp; provenance</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>Original title</dt><dd>Impact of ChatGPT on ESL students&#x27; academic writing skills</dd></div><div class="source-detail-row"><dt>Year</dt><dd>2024</dd></div><div class="source-detail-row"><dt>Authority</dt><dd>Tier 1 DOI-verified paper</dd></div><div class="source-detail-row"><dt>Source location</dt><dd>https://link.springer.com/article/10.1186/s40561-024-00295-9</dd></div><div class="source-detail-row"><dt>Verifiable link</dt><dd><a href="https://link.springer.com/article/10.1186/s40561-024-00295-9">https://link.springer.com/article/10.1186/s40561-024-00295-9</a></dd></div></dl></details></td><td>2024</td><td>Tier 1 DOI-verified paper</td><td class='cell-main'><a href='https://link.springer.com/article/10.1186/s40561-024-00295-9'>https://link.springer.com/article/10.1186/s40561-024-00295-9</a></td></tr><tr><td><code>S-2023-peng</code></td><td class='cell-main source-title-cell'>The Impact of AI on Developer Productivity: Evidence from GitHub Copilot <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>View source &amp; provenance</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>Original title</dt><dd>The Impact of AI on Developer Productivity: Evidence from GitHub Copilot</dd></div><div class="source-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="source-detail-row"><dt>Authority</dt><dd>tier2_academic_database</dd></div><div class="source-detail-row"><dt>Source location</dt><dd>https://arxiv.org/abs/2302.06590</dd></div><div class="source-detail-row"><dt>Verifiable link</dt><dd><a href="https://doi.org/10.48550/arXiv.2302.06590">https://doi.org/10.48550/arXiv.2302.06590</a></dd></div></dl></details></td><td>2023</td><td>tier2_academic_database</td><td class='cell-main'><a href='https://doi.org/10.48550/arXiv.2302.06590'>https://doi.org/10.48550/arXiv.2302.06590</a></td></tr><tr><td><code>S-2023-yetistiren</code></td><td class='cell-main source-title-cell'>GitHub Copilot AI Pair Programmer: Asset or Liability? <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>View source &amp; provenance</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>Original title</dt><dd>GitHub Copilot AI Pair Programmer: Asset or Liability?</dd></div><div class="source-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="source-detail-row"><dt>Authority</dt><dd>Tier 1 DOI-verified paper</dd></div><div class="source-detail-row"><dt>Source location</dt><dd>https://doi.org/10.1016/j.jss.2023.111734</dd></div><div class="source-detail-row"><dt>Verifiable link</dt><dd><a href="https://doi.org/10.1016/j.jss.2023.111734">https://doi.org/10.1016/j.jss.2023.111734</a></dd></div></dl></details></td><td>2023</td><td>Tier 1 DOI-verified paper</td><td class='cell-main'><a href='https://doi.org/10.1016/j.jss.2023.111734'>https://doi.org/10.1016/j.jss.2023.111734</a></td></tr><tr><td><code>S-2022-finnie-ansley</code></td><td class='cell-main source-title-cell'>Using GitHub Copilot to Solve Introductory Programming Problems <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>View source &amp; provenance</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>Original title</dt><dd>Using GitHub Copilot to Solve Introductory Programming Problems</dd></div><div class="source-detail-row"><dt>Year</dt><dd>2022</dd></div><div class="source-detail-row"><dt>Authority</dt><dd>Tier 1 DOI-verified paper</dd></div><div class="source-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3545945.3569830</dd></div><div class="source-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3545945.3569830">https://dl.acm.org/doi/10.1145/3545945.3569830</a></dd></div></dl></details></td><td>2022</td><td>Tier 1 DOI-verified paper</td><td class='cell-main'><a href='https://dl.acm.org/doi/10.1145/3545945.3569830'>https://dl.acm.org/doi/10.1145/3545945.3569830</a></td></tr><tr><td><code>S-2023-explanations-compare</code></td><td class='cell-main source-title-cell'>Comparing Code Explanations Created by Students and Large Language Models <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>View source &amp; provenance</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>Original title</dt><dd>Comparing Code Explanations Created by Students and Large Language Models</dd></div><div class="source-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="source-detail-row"><dt>Authority</dt><dd>Tier 1 DOI-verified paper</dd></div><div class="source-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3587102.3588785</dd></div><div class="source-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3587102.3588785">https://dl.acm.org/doi/10.1145/3587102.3588785</a></dd></div></dl></details></td><td>2023</td><td>Tier 1 DOI-verified paper</td><td class='cell-main'><a href='https://dl.acm.org/doi/10.1145/3587102.3588785'>https://dl.acm.org/doi/10.1145/3587102.3588785</a></td></tr><tr><td><code>S-2022-vaithilingam</code></td><td class='cell-main source-title-cell'>Expectation vs. Experience: Evaluating the Usability of Code Generation Tools <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>View source &amp; provenance</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>Original title</dt><dd>Expectation vs. Experience: Evaluating the Usability of Code Generation Tools</dd></div><div class="source-detail-row"><dt>Year</dt><dd>2022</dd></div><div class="source-detail-row"><dt>Authority</dt><dd>Tier 1 DOI-verified paper</dd></div><div class="source-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3491101.3519665</dd></div><div class="source-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3491101.3519665">https://dl.acm.org/doi/10.1145/3491101.3519665</a></dd></div></dl></details></td><td>2022</td><td>Tier 1 DOI-verified paper</td><td class='cell-main'><a href='https://dl.acm.org/doi/10.1145/3491101.3519665'>https://dl.acm.org/doi/10.1145/3491101.3519665</a></td></tr></tbody></table></div><h3>Fetch provenance</h3><p class="provenance-summary">Search provider: n/a</p><p class='provenance-empty'>No per-source fetch records (sources provided directly by the research pipeline).</p></div></section></main></div>
2544
+ <svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="Evaluation design flow"><title>Evaluation design flow</title><desc>Evaluation flow across baseline, post test, retention and transfer; full metrics and analysis plan are in the evaluation section.</desc><defs><marker id="arr-en-full-evaluation" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">Evaluation Design Flow</text><rect x="30" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="105.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="105.0" y="138.0">Baseline</tspan></text><text x="105.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">pre-test</text><line x1="180" y1="138" x2="192" y2="138" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-evaluation)"/><rect x="192" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="267.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="267.0" y="138.0">Post test</tspan></text><text x="267.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">post-test</text><line x1="342" y1="138" x2="354" y2="138" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-evaluation)"/><rect x="354" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="429.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="429.0" y="138.0">Retention</tspan></text><text x="429.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">retention test</text><line x1="504" y1="138" x2="516" y2="138" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-evaluation)"/><rect x="516" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="591.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="591.0" y="138.0">Transfer</tspan></text><text x="591.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">transfer test (no AI)</text></svg><div class="visual-suppressed"><strong>Benchmark visual suppressed</strong><p>result.json carries no benchmark.baselines, so this visual is omitted; see the standalone benchmark report.</p></div></div></section><section id="full-06-sources-en" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>06 Sources, Traceability &amp; Appendix</h2><p class="full-chapter-lead">Preserve original sources, URLs, evidence IDs and retrieval metadata for auditability.</p></header><div class="full-chapter-body"><h3>Source list</h3><div class='table-wrap'><table class='data-table source-table'><thead><tr><th>ID</th><th>Title</th><th>Year</th><th>Authority</th><th>Verifiable location</th></tr></thead><tbody><tr><td><code>S-2023-kazemitabaar</code></td><td class='cell-main source-title-cell'>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>View source &amp; provenance</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>Original title</dt><dd>Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming</dd></div><div class="source-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="source-detail-row"><dt>Authority</dt><dd>Tier 1 DOI-verified paper</dd></div><div class="source-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3544548.3580919</dd></div><div class="source-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3544548.3580919">https://dl.acm.org/doi/10.1145/3544548.3580919</a></dd></div></dl></details></td><td>2023</td><td>Tier 1 DOI-verified paper</td><td class='cell-main'><a href='https://dl.acm.org/doi/10.1145/3544548.3580919'>https://dl.acm.org/doi/10.1145/3544548.3580919</a></td></tr><tr><td><code>S-2025-bastani</code></td><td class='cell-main source-title-cell'>Generative AI without guardrails can harm learning: Evidence from high school mathematics <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>View source &amp; provenance</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>Original title</dt><dd>Generative AI without guardrails can harm learning: Evidence from high school mathematics</dd></div><div class="source-detail-row"><dt>Year</dt><dd>2025</dd></div><div class="source-detail-row"><dt>Authority</dt><dd>Tier 1 DOI-verified paper</dd></div><div class="source-detail-row"><dt>Source location</dt><dd>https://www.pnas.org/doi/10.1073/pnas.2422633122</dd></div><div class="source-detail-row"><dt>Verifiable link</dt><dd><a href="https://www.pnas.org/doi/10.1073/pnas.2422633122">https://www.pnas.org/doi/10.1073/pnas.2422633122</a></dd></div></dl></details></td><td>2025</td><td>Tier 1 DOI-verified paper</td><td class='cell-main'><a href='https://www.pnas.org/doi/10.1073/pnas.2422633122'>https://www.pnas.org/doi/10.1073/pnas.2422633122</a></td></tr><tr><td><code>S-2024-marzuki</code></td><td class='cell-main source-title-cell'>Impact of ChatGPT on ESL students&#x27; academic writing skills <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>View source &amp; provenance</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>Original title</dt><dd>Impact of ChatGPT on ESL students&#x27; academic writing skills</dd></div><div class="source-detail-row"><dt>Year</dt><dd>2024</dd></div><div class="source-detail-row"><dt>Authority</dt><dd>Tier 1 DOI-verified paper</dd></div><div class="source-detail-row"><dt>Source location</dt><dd>https://link.springer.com/article/10.1186/s40561-024-00295-9</dd></div><div class="source-detail-row"><dt>Verifiable link</dt><dd><a href="https://link.springer.com/article/10.1186/s40561-024-00295-9">https://link.springer.com/article/10.1186/s40561-024-00295-9</a></dd></div></dl></details></td><td>2024</td><td>Tier 1 DOI-verified paper</td><td class='cell-main'><a href='https://link.springer.com/article/10.1186/s40561-024-00295-9'>https://link.springer.com/article/10.1186/s40561-024-00295-9</a></td></tr><tr><td><code>S-2023-peng</code></td><td class='cell-main source-title-cell'>The Impact of AI on Developer Productivity: Evidence from GitHub Copilot <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>View source &amp; provenance</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>Original title</dt><dd>The Impact of AI on Developer Productivity: Evidence from GitHub Copilot</dd></div><div class="source-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="source-detail-row"><dt>Authority</dt><dd>Tier2 Academic Database</dd></div><div class="source-detail-row"><dt>Source location</dt><dd>https://arxiv.org/abs/2302.06590</dd></div><div class="source-detail-row"><dt>Verifiable link</dt><dd><a href="https://doi.org/10.48550/arXiv.2302.06590">https://doi.org/10.48550/arXiv.2302.06590</a></dd></div></dl></details></td><td>2023</td><td>Tier2 Academic Database</td><td class='cell-main'><a href='https://doi.org/10.48550/arXiv.2302.06590'>https://doi.org/10.48550/arXiv.2302.06590</a></td></tr><tr><td><code>S-2023-yetistiren</code></td><td class='cell-main source-title-cell'>GitHub Copilot AI Pair Programmer: Asset or Liability? <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>View source &amp; provenance</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>Original title</dt><dd>GitHub Copilot AI Pair Programmer: Asset or Liability?</dd></div><div class="source-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="source-detail-row"><dt>Authority</dt><dd>Tier 1 DOI-verified paper</dd></div><div class="source-detail-row"><dt>Source location</dt><dd>https://doi.org/10.1016/j.jss.2023.111734</dd></div><div class="source-detail-row"><dt>Verifiable link</dt><dd><a href="https://doi.org/10.1016/j.jss.2023.111734">https://doi.org/10.1016/j.jss.2023.111734</a></dd></div></dl></details></td><td>2023</td><td>Tier 1 DOI-verified paper</td><td class='cell-main'><a href='https://doi.org/10.1016/j.jss.2023.111734'>https://doi.org/10.1016/j.jss.2023.111734</a></td></tr><tr><td><code>S-2022-finnie-ansley</code></td><td class='cell-main source-title-cell'>Using GitHub Copilot to Solve Introductory Programming Problems <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>View source &amp; provenance</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>Original title</dt><dd>Using GitHub Copilot to Solve Introductory Programming Problems</dd></div><div class="source-detail-row"><dt>Year</dt><dd>2022</dd></div><div class="source-detail-row"><dt>Authority</dt><dd>Tier 1 DOI-verified paper</dd></div><div class="source-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3545945.3569830</dd></div><div class="source-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3545945.3569830">https://dl.acm.org/doi/10.1145/3545945.3569830</a></dd></div></dl></details></td><td>2022</td><td>Tier 1 DOI-verified paper</td><td class='cell-main'><a href='https://dl.acm.org/doi/10.1145/3545945.3569830'>https://dl.acm.org/doi/10.1145/3545945.3569830</a></td></tr><tr><td><code>S-2023-explanations-compare</code></td><td class='cell-main source-title-cell'>Comparing Code Explanations Created by Students and Large Language Models <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>View source &amp; provenance</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>Original title</dt><dd>Comparing Code Explanations Created by Students and Large Language Models</dd></div><div class="source-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="source-detail-row"><dt>Authority</dt><dd>Tier 1 DOI-verified paper</dd></div><div class="source-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3587102.3588785</dd></div><div class="source-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3587102.3588785">https://dl.acm.org/doi/10.1145/3587102.3588785</a></dd></div></dl></details></td><td>2023</td><td>Tier 1 DOI-verified paper</td><td class='cell-main'><a href='https://dl.acm.org/doi/10.1145/3587102.3588785'>https://dl.acm.org/doi/10.1145/3587102.3588785</a></td></tr><tr><td><code>S-2022-vaithilingam</code></td><td class='cell-main source-title-cell'>Expectation vs. Experience: Evaluating the Usability of Code Generation Tools <span class="cite-badge cite-ok">DOI ✓</span><details class="source-expander detail-expander"><summary>View source &amp; provenance</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>Original title</dt><dd>Expectation vs. Experience: Evaluating the Usability of Code Generation Tools</dd></div><div class="source-detail-row"><dt>Year</dt><dd>2022</dd></div><div class="source-detail-row"><dt>Authority</dt><dd>Tier 1 DOI-verified paper</dd></div><div class="source-detail-row"><dt>Source location</dt><dd>https://dl.acm.org/doi/10.1145/3491101.3519665</dd></div><div class="source-detail-row"><dt>Verifiable link</dt><dd><a href="https://dl.acm.org/doi/10.1145/3491101.3519665">https://dl.acm.org/doi/10.1145/3491101.3519665</a></dd></div></dl></details></td><td>2022</td><td>Tier 1 DOI-verified paper</td><td class='cell-main'><a href='https://dl.acm.org/doi/10.1145/3491101.3519665'>https://dl.acm.org/doi/10.1145/3491101.3519665</a></td></tr></tbody></table></div><h3>Fetch provenance</h3><p class="provenance-summary">Search provider: n/a</p><p class='provenance-empty'>No per-source fetch records (sources provided directly by the research pipeline).</p></div></section></main></div>
2225
2545
  </div>
2226
2546
  <footer class="report-footer"><p>EduEvidence Evidence Report · Schema PASS · Claim Binding PASS · Numeric Consistency PASS · Bilingual Structure PASS · Human Language PASS · False Precision PASS · Lieflat Data Bound PASS · Axis Distortion NOT_CHECKED · Colorblind Safe NOT_CHECKED · single-file offline · source: result.json</p></footer>
2227
2547
  </div>