eduevidence 5.2.0 → 6.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (276) hide show
  1. package/README.md +54 -42
  2. package/README.zh-CN.md +51 -28
  3. package/SKILL.md +390 -133
  4. package/agents/openai.yaml +4 -0
  5. package/assets/readme/controlled-execution.svg +34 -0
  6. package/assets/readme/logo.png +0 -0
  7. package/assets/readme/research-workflow.svg +56 -0
  8. package/assets/readme/studio-graph.png +0 -0
  9. package/assets/readme/studio-overview.png +0 -0
  10. package/assets/readme/studio-reports.png +0 -0
  11. package/autoevolve/config.yaml +17 -0
  12. package/autoevolve/program.md +25 -0
  13. package/autoevolve/protected.manifest.yaml +34 -0
  14. package/benchmarks/adversarial/cases.jsonl +7 -0
  15. package/benchmarks/evidence-library.json +5268 -0
  16. package/benchmarks/partitions.json +8 -0
  17. package/docs/architecture.md +220 -0
  18. package/docs/autoresearch-evolution-plan.md +2903 -0
  19. package/docs/autoresearch-implementation-status.md +101 -0
  20. package/docs/demo-storyboard.md +20 -0
  21. package/docs/demo-workplace-ai.md +92 -0
  22. package/docs/demo.md +32 -0
  23. package/docs/install-guide.md +150 -0
  24. package/docs/orchestration-role-model.md +1254 -0
  25. package/docs/release-closeout/README.md +17 -0
  26. package/docs/release-closeout/frontend-acceptance.md +23 -0
  27. package/docs/release-closeout/issues.md +19 -0
  28. package/docs/release-closeout/verification.md +28 -0
  29. package/docs/release-contract.md +108 -0
  30. package/docs/research-studio-guide.zh-CN.md +166 -0
  31. package/eduevidence_cli.py +17 -11
  32. package/engine/_resources.py +13 -0
  33. package/engine/autoevolve/__init__.py +3 -0
  34. package/engine/autoevolve/agent_view.py +167 -0
  35. package/engine/autoevolve/core.py +357 -0
  36. package/engine/autoevolve/events.py +11 -0
  37. package/engine/autoevolve/git_workspace.py +77 -0
  38. package/engine/autoevolve/projection.py +23 -0
  39. package/engine/autoevolve/runner.py +413 -0
  40. package/engine/autoevolve/trust.py +146 -0
  41. package/engine/autoresearch/__init__.py +6 -0
  42. package/engine/autoresearch/commit.py +132 -0
  43. package/engine/autoresearch/contracts.py +126 -0
  44. package/engine/autoresearch/controller.py +207 -0
  45. package/engine/autoresearch/events.py +12 -0
  46. package/engine/autoresearch/gap_priority.py +168 -0
  47. package/engine/autoresearch/projection.py +30 -0
  48. package/engine/autoresearch/research_memory.py +59 -0
  49. package/engine/autoresearch/saturation.py +91 -0
  50. package/engine/briefs.py +2 -1
  51. package/engine/capabilities.py +1 -0
  52. package/engine/contracts.py +3 -1
  53. package/engine/evidencecore.py +7 -5
  54. package/engine/gaps.py +90 -51
  55. package/engine/judge_pack.py +65 -0
  56. package/engine/library_builtin.py +3 -1
  57. package/engine/living.py +2 -1
  58. package/engine/meta_synthesis.py +3 -1
  59. package/engine/orchestration.py +460 -0
  60. package/engine/pilot.py +2 -1
  61. package/engine/project.py +2 -2
  62. package/engine/research_service.py +113 -0
  63. package/engine/studio_read_model.py +400 -0
  64. package/engine/tribunal.py +1 -2
  65. package/engine/update.py +1 -0
  66. package/engine/versions.py +1 -1
  67. package/engine/worker_result.py +109 -0
  68. package/engine/workflows.py +70 -0
  69. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1720 -0
  70. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
  71. package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
  72. package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
  73. package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
  74. package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
  75. package/examples/ai-coding-assistant-evidence/frame.json +48 -0
  76. package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
  77. package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
  78. package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
  79. package/examples/ai-coding-assistant-evidence/report_spec.json +219 -0
  80. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2614 -0
  81. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2614 -0
  82. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2614 -0
  83. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2614 -0
  84. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2614 -0
  85. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1720 -0
  86. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1720 -0
  87. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1720 -0
  88. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1720 -0
  89. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1720 -0
  90. package/examples/ai-coding-assistant-evidence/result.json +1453 -0
  91. package/examples/ai-coding-assistant-evidence/result.zh.json +1453 -0
  92. package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
  93. package/examples/ai-coding-assistant-evidence/verdict.json +103 -0
  94. package/examples/workplace-ai-assistant/claims.jsonl +4 -0
  95. package/examples/workplace-ai-assistant/evaluation.json +19 -0
  96. package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
  97. package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
  98. package/examples/workplace-ai-assistant/frame.json +41 -0
  99. package/examples/workplace-ai-assistant/intervention.json +27 -0
  100. package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
  101. package/examples/workplace-ai-assistant/methodology.json +60 -0
  102. package/examples/workplace-ai-assistant/report_spec.json +55 -0
  103. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2484 -0
  104. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2484 -0
  105. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2484 -0
  106. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2484 -0
  107. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2484 -0
  108. package/examples/workplace-ai-assistant/result.json +553 -0
  109. package/examples/workplace-ai-assistant/result.zh.json +553 -0
  110. package/examples/workplace-ai-assistant/search_log.json +19 -0
  111. package/examples/workplace-ai-assistant/sources.jsonl +3 -0
  112. package/examples/workplace-ai-assistant/validation_result.json +9 -0
  113. package/examples/workplace-ai-assistant/verdict.json +52 -0
  114. package/install.sh +7 -7
  115. package/integrations/orchestration_dispatch.py +146 -0
  116. package/package.json +37 -3
  117. package/pyproject.toml +11 -20
  118. package/references/autoresearch.md +30 -0
  119. package/references/evaluation-policy.md +24 -0
  120. package/references/orchestration.md +22 -0
  121. package/references/scientific-invariants.md +19 -0
  122. package/retrieval/audit.py +154 -0
  123. package/schemas/intervention.schema.json +106 -21
  124. package/schemas/report-result.schema.json +9 -1
  125. package/schemas/v2/project.schema.json +2 -2
  126. package/schemas/v2/run.schema.json +1 -1
  127. package/schemas/vNext/autoevolve-session.schema.json +1 -0
  128. package/schemas/vNext/eval-snapshot.schema.json +1 -0
  129. package/schemas/vNext/execution-plan.schema.json +1 -0
  130. package/schemas/vNext/gap-priority.schema.json +1 -0
  131. package/schemas/vNext/negative-search-record.schema.json +1 -0
  132. package/schemas/vNext/research-iteration.schema.json +1 -0
  133. package/schemas/vNext/research-strategy.schema.json +1 -0
  134. package/schemas/vNext/skill-experiment.schema.json +1 -0
  135. package/schemas/vNext/task-spec.schema.json +1 -0
  136. package/schemas/vNext/worker-result.schema.json +1 -0
  137. package/scripts/benchmark_judge.py +2 -2
  138. package/scripts/benchmark_v3.py +26 -43
  139. package/scripts/build_esl_artifacts.py +2 -2
  140. package/scripts/build_evidence_library.py +2 -2
  141. package/scripts/build_gh_pages.py +98 -0
  142. package/scripts/build_readme_diagrams.py +72 -0
  143. package/scripts/build_report_variants.py +85 -0
  144. package/scripts/check_autoresearch_invariants.py +95 -0
  145. package/scripts/daily_evolve.py +30 -0
  146. package/scripts/dashboard_server.py +130 -101
  147. package/scripts/did_regression.py +5 -30
  148. package/scripts/enrich_projects_human_and_lieflat.py +1 -1
  149. package/scripts/generate_metrics.py +4 -3
  150. package/scripts/generate_new_projects.py +1 -1
  151. package/scripts/orchestrator.py +172 -18
  152. package/scripts/rebake_all_5themes.py +1 -2
  153. package/scripts/research_auto_cli.py +475 -0
  154. package/scripts/run_workspace.py +17 -7
  155. package/scripts/search_provenance.py +64 -0
  156. package/scripts/serve_web.py +9 -10
  157. package/scripts/skill_lint.py +1 -1
  158. package/scripts/skill_payload.py +78 -0
  159. package/scripts/validate_schema.py +15 -1
  160. package/scripts/vnext_cli.py +133 -0
  161. package/setup.py +12 -0
  162. package/skill/roles/registry.yaml +45 -0
  163. package/skill/sub-skills/report-generation/SKILL.md +12 -6
  164. package/skill/task-briefs/applicability.md +3 -0
  165. package/skill/task-briefs/projection.md +3 -0
  166. package/skill/workflows/decision-and-pilot.md +10 -0
  167. package/skill/workflows/evaluate-and-update.md +10 -0
  168. package/skill/workflows/evidence-review.md +13 -0
  169. package/visualization/eduevidence-report/assets/base.css +2 -2
  170. package/visualization/eduevidence-report/assets/reader.css +752 -0
  171. package/visualization/eduevidence-report/assets/reader.js +132 -0
  172. package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
  173. package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
  174. package/visualization/eduevidence-report/scripts/build_report.py +58 -65
  175. package/visualization/eduevidence-report/scripts/lieflat_engine.py +24 -100
  176. package/visualization/eduevidence-report/themes/academic.css +1 -1
  177. package/visualization/eduevidence-report/themes/claude.css +1 -1
  178. package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
  179. package/visualization/eduevidence-report/themes/datalab.css +2 -2
  180. package/visualization/eduevidence-report/themes/presentation.css +2 -2
  181. package/web/README.md +18 -0
  182. package/web/index.html +53 -0
  183. package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
  184. package/web/studio/assets/index-CzXocaGv.css +1 -0
  185. package/web/studio/assets/index-pa7jD7n4.js +230 -0
  186. package/web/studio/config.json +1 -0
  187. package/web/studio/index.html +14 -0
  188. package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
  189. package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
  190. package/engine/__pycache__/bias.cpython-312.pyc +0 -0
  191. package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
  192. package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
  193. package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
  194. package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
  195. package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
  196. package/engine/__pycache__/events.cpython-312.pyc +0 -0
  197. package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
  198. package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
  199. package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
  200. package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
  201. package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
  202. package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
  203. package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
  204. package/engine/__pycache__/ids.cpython-312.pyc +0 -0
  205. package/engine/__pycache__/library.cpython-312.pyc +0 -0
  206. package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
  207. package/engine/__pycache__/living.cpython-312.pyc +0 -0
  208. package/engine/__pycache__/log.cpython-312.pyc +0 -0
  209. package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
  210. package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
  211. package/engine/__pycache__/migration.cpython-312.pyc +0 -0
  212. package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
  213. package/engine/__pycache__/paths.cpython-312.pyc +0 -0
  214. package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
  215. package/engine/__pycache__/planner.cpython-312.pyc +0 -0
  216. package/engine/__pycache__/project.cpython-312.pyc +0 -0
  217. package/engine/__pycache__/projections.cpython-312.pyc +0 -0
  218. package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
  219. package/engine/__pycache__/run.cpython-312.pyc +0 -0
  220. package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
  221. package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
  222. package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
  223. package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
  224. package/engine/__pycache__/update.cpython-312.pyc +0 -0
  225. package/engine/__pycache__/versions.cpython-312.pyc +0 -0
  226. package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
  227. package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
  228. package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
  229. package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
  230. package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
  231. package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
  232. package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
  233. package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
  234. package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
  235. package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
  236. package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
  237. package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
  238. package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
  239. package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
  240. package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
  241. package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
  242. package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
  243. package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
  244. package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
  245. package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
  246. package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
  247. package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
  248. package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
  249. package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
  250. package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
  251. package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
  252. package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
  253. package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
  254. package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
  255. package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
  256. package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
  257. package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
  258. package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
  259. package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
  260. package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
  261. package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
  262. package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
  263. package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
  264. package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
  265. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
  266. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
  267. package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
  268. package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
  269. package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
  270. package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
  271. package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
  272. package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
  273. package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
  274. package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
  275. package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
  276. package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
@@ -0,0 +1,8 @@
1
+ {"source_id": "S-2023-kazemitabaar", "title": "Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming", "year": 2023, "doi": "10.1145/3544548.3580919", "canonical_url": "https://dl.acm.org/doi/10.1145/3544548.3580919", "authority_level": "tier1_paper_doi", "source_location": "https://dl.acm.org/doi/10.1145/3544548.3580919", "doi_verified": true, "retracted": false}
2
+ {"source_id": "S-2025-bastani", "title": "Generative AI without guardrails can harm learning: Evidence from high school mathematics", "year": 2025, "doi": "10.1073/pnas.2422633122", "canonical_url": "https://www.pnas.org/doi/10.1073/pnas.2422633122", "authority_level": "tier1_paper_doi", "source_location": "https://www.pnas.org/doi/10.1073/pnas.2422633122", "doi_verified": true, "retracted": false}
3
+ {"source_id": "S-2024-marzuki", "title": "Impact of ChatGPT on ESL students' academic writing skills", "year": 2024, "doi": "10.1186/s40561-024-00295-9", "canonical_url": "https://link.springer.com/article/10.1186/s40561-024-00295-9", "authority_level": "tier1_paper_doi", "source_location": "https://link.springer.com/article/10.1186/s40561-024-00295-9", "doi_verified": true, "retracted": false}
4
+ {"source_id": "S-2023-peng", "title": "The Impact of AI on Developer Productivity: Evidence from GitHub Copilot", "year": 2023, "doi": "10.48550/arXiv.2302.06590", "canonical_url": "https://doi.org/10.48550/arXiv.2302.06590", "authority_level": "tier2_academic_database", "source_location": "https://arxiv.org/abs/2302.06590", "doi_verified": true, "retracted": false}
5
+ {"source_id": "S-2023-yetistiren", "title": "GitHub Copilot AI Pair Programmer: Asset or Liability?", "year": 2023, "doi": "10.1016/j.jss.2023.111734", "canonical_url": "https://doi.org/10.1016/j.jss.2023.111734", "authority_level": "tier1_paper_doi", "source_location": "https://doi.org/10.1016/j.jss.2023.111734", "doi_verified": true, "retracted": false}
6
+ {"source_id": "S-2022-finnie-ansley", "title": "Using GitHub Copilot to Solve Introductory Programming Problems", "year": 2022, "doi": "10.1145/3545945.3569830", "canonical_url": "https://dl.acm.org/doi/10.1145/3545945.3569830", "authority_level": "tier1_paper_doi", "source_location": "https://dl.acm.org/doi/10.1145/3545945.3569830", "doi_verified": true, "retracted": false}
7
+ {"source_id": "S-2023-explanations-compare", "title": "Comparing Code Explanations Created by Students and Large Language Models", "year": 2023, "doi": "10.1145/3587102.3588785", "canonical_url": "https://dl.acm.org/doi/10.1145/3587102.3588785", "authority_level": "tier1_paper_doi", "source_location": "https://dl.acm.org/doi/10.1145/3587102.3588785", "doi_verified": true, "retracted": false}
8
+ {"source_id": "S-2022-vaithilingam", "title": "Expectation vs. Experience: Evaluating the Usability of Code Generation Tools", "year": 2022, "doi": "10.1145/3491101.3519665", "canonical_url": "https://dl.acm.org/doi/10.1145/3491101.3519665", "authority_level": "tier1_paper_doi", "source_location": "https://dl.acm.org/doi/10.1145/3491101.3519665", "doi_verified": true, "retracted": false}
@@ -0,0 +1,103 @@
1
+ {
2
+ "decision_question": "大一 C 语言课程是否应该允许学生使用生成式 AI 编程助手?",
3
+ "target_population": "university first-year computer science students learning C programming for the first time",
4
+ "target_context": "16-week lecture-lab course, 60 students, TA support, offline",
5
+ "supported_claims": [
6
+ "AI coding assistants reliably increase task performance during training (completion speed, correctness) — E-001, E-006.",
7
+ "Unguarded generative AI access can harm independent problem solving when access is removed — E-004.",
8
+ "Guardrail design (hints instead of answers) substantially mitigates the negative learning effect — E-005.",
9
+ "Task performance gains do not automatically imply learning gains — E-004 vs E-006 (within-study contrast).",
10
+ "Tool capability is substantial: Codex solves roughly half to three-quarters of CS1 exam-style questions — E-010.",
11
+ "Professional-developer RCT shows ~55% faster task completion with Copilot; directness limited by professional population — E-008.",
12
+ "LLM code explanations rate comparable to student-authored explanations, viable as scaffold material — E-011."
13
+ ],
14
+ "uncertain_claims": [
15
+ "Whether AI coding assistants improve or preserve actual programming learning in university novices — no direct university-level RCT in reviewed set [无直接证据]",
16
+ "Whether one-week neutral retention (Kazemitabaar 2023) extends to a semester — E-003.",
17
+ "Whether benchmark quality findings (E-009) and explanation-quality ratings (E-011) translate into classroom learning gains.",
18
+ "How comprehension/ownership difficulties documented in usability studies (E-012) behave over a full semester with guardrails."
19
+ ],
20
+ "contradicted_claims": [
21
+ "The claim 'AI tools always improve learning' is contradicted by E-004 (unguarded access, -17% independent exam).",
22
+ "The claim 'speed gains equal learning gains' is contradicted by the task-vs-learning separation across E-001/E-006/E-008 vs E-004."
23
+ ],
24
+ "reason_for_disagreement": "Disagreement comes from outcome separation (task vs learning), tool design (guarded vs unguarded), and population (K-12 / professionals vs university). Task-performance evidence is consistently positive across randomized and benchmark studies; the only study measuring independent performance after AI removal shows harm without guardrails; usability and artifact studies add dependence and quality caveats rather than resolving the learning question.",
25
+ "methodology_summary": "Eight real sources: three randomized experiments (Kazemitabaar 2023 n=69 K-12; Bastani 2025 n≈950 high-school mathematics; Peng 2023 n=95 professional developers, preprint), one ESL writing mixed-methods study (Marzuki 2024), plus benchmark/capability/usability studies (Yetistiren 2023; Finnie-Ansley 2022; explanation-comparison 2023; Vaithilingam 2022). No direct RCT in university programming courses. Internal validity of the core RCTs is strong; directness to first-year university C programming is weak. All sources carry registry-verified DOIs (see benchmarks/doi-audit/report.md).",
26
+ "outcome_specific_findings": {
27
+ "completion_time": "positive during training and professional tasks (E-001, E-008)",
28
+ "independent_problem_solving": "neutral-to-negative without guardrails (E-002, E-004)",
29
+ "retention": "neutral over 1 week (E-003)",
30
+ "assignment_score": "positive during practice, negative on closed-book exam (E-004, E-006); tool itself scores passing-level on CS1 questions (E-010)",
31
+ "code_quality": "mixed on benchmarks; security concerns documented (E-009)",
32
+ "metacognition": "LLM explanations compare well (E-011) while novice ownership/debugging difficulties persist (E-012)",
33
+ "ai_dependency": "documented crutch behavior with unguarded tool (E-004, E-005, E-012)"
34
+ },
35
+ "short_term_effect": "Task performance reliably increases; learning effect null-to-negative without guardrails.",
36
+ "long_term_effect": "No evidence beyond one week; long-term learning effect unknown.",
37
+ "transfer_effect": "No full transfer evidence; manual code-modification not harmed in one small study (E-002).",
38
+ "risk_effect": "AI dependency and over-reliance risk is real and documented for unguarded usage (E-004) and foreshadowed by usability findings (E-012).",
39
+ "applicability": {
40
+ "suitable_for": "pilot in first-year C course with guardrailed usage policy",
41
+ "not_suitable_for": "unrestricted AI adoption without usage policy",
42
+ "required_conditions": [
43
+ "guardrailed AI usage policy (hints not answers, modeled on GPT Tutor arm)",
44
+ "no-AI transfer assessment",
45
+ "TA support"
46
+ ]
47
+ },
48
+ "confidence": "Moderate",
49
+ "confidence_breakdown": {
50
+ "score": 0.586,
51
+ "evidence_quality": 0.758,
52
+ "consistency": 0.667,
53
+ "directness": 0.458,
54
+ "evidence_count": 12,
55
+ "independent_studies": 8,
56
+ "independent_samples": 8,
57
+ "count_term": 1.0,
58
+ "conflict_penalty": 0.15,
59
+ "unsupported_penalty": 0.0
60
+ },
61
+ "what_can_be_claimed": [
62
+ "AI coding assistants raise task performance for novices during training.",
63
+ "Unguarded access carries a real risk of hurting independent problem solving.",
64
+ "Guardrail design can mitigate that risk.",
65
+ "Direct evidence for university C programming learning is missing.",
66
+ "Tool capability headroom is large (CS1 question pass rates; professional speed RCT)."
67
+ ],
68
+ "what_cannot_be_claimed": [
69
+ "AI coding assistants improve (or even preserve) university students' programming learning.",
70
+ "Any long-term or retention benefit.",
71
+ "Any claim about which students benefit, based on university samples.",
72
+ "That benchmark or usability findings substitute for classroom learning outcomes."
73
+ ],
74
+ "missing_evidence": [
75
+ "RCT of AI coding assistants in university programming courses with retention and no-AI transfer tests.",
76
+ "Studies varying AI usage policy within the same course.",
77
+ "Longitudinal data on AI dependency beyond one course.",
78
+ "Peer-reviewed replication of the professional speed RCT (Peng et al. remains a preprint)."
79
+ ],
80
+ "recommended_action": "pilot",
81
+ "decision_rationale": "Positive task-performance evidence plus documented unguarded-access risk, mixed quality/usability signals, and missing university-level learning evidence → bounded, guardrailed pilot with evaluation, not full adoption.",
82
+ "exceeds_evidence_boundary": [
83
+ "Claiming 'AI coding assistants improve learning' exceeds the boundary: direct learning-effect evidence is missing.",
84
+ "Claiming 'AI works for everyone' exceeds the boundary: population and subject mismatch."
85
+ ],
86
+ "confidence_score": 0.586,
87
+ "confidence_policy_version": "2026-08-12.v2",
88
+ "independent_studies": 8,
89
+ "independent_samples": 8,
90
+ "raw_model_confidence": "Moderate",
91
+ "raw_model_confidence_breakdown": {
92
+ "score": 0.5,
93
+ "evidence_quality": 0.7,
94
+ "consistency": 0.6,
95
+ "directness": 0.4,
96
+ "evidence_count": 12,
97
+ "independent_studies": 8,
98
+ "independent_samples": 8,
99
+ "count_term": 1.0,
100
+ "conflict_penalty": 0.0,
101
+ "unsupported_penalty": 0.0
102
+ }
103
+ }
@@ -0,0 +1,4 @@
1
+ {"claim_id": "C-001", "claim": "AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.", "outcome_type": "completion_time", "evidence_ids": ["E-001"], "status": "SUPPORTED", "pooled_effect_g": null}
2
+ {"claim_id": "C-002", "claim": "ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.", "outcome_type": "completion_time", "evidence_ids": ["E-002"], "status": "SUPPORTED", "pooled_effect_g": null}
3
+ {"claim_id": "C-003", "claim": "AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.", "outcome_type": "accuracy", "evidence_ids": ["E-003"], "status": "SUPPORTED", "pooled_effect_g": null}
4
+ {"claim_id": "C-004", "claim": "Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits.", "outcome_type": "accuracy", "evidence_ids": ["E-004"], "status": "SUPPORTED", "pooled_effect_g": null}
@@ -0,0 +1,19 @@
1
+ {
2
+ "research_question": "Should an enterprise customer-support team introduce a generative AI assistant?",
3
+ "groups": {
4
+ "treatment": "Eligible teams randomly assigned to supervised assistant access, stratified by tenure and baseline performance.",
5
+ "comparison": "Concurrent teams retaining the existing workflow; document contamination."
6
+ },
7
+ "baseline": "Record resolution rate, paid hours, repeat contacts, blinded quality and review costs before allocation.",
8
+ "post_test": "Assess the same outcomes at pilot end; separately audit privacy and unsafe commitments.",
9
+ "analysis_plan": "Preregister an intention-to-treat comparison with team-clustered uncertainty and subgroup estimates. Set sample size and noninferiority margins from local baseline variance and operational priorities before enrollment.",
10
+ "success_threshold": "Expand only if quality is noninferior, verified resolutions per paid hour improve, net cost is acceptable and no serious unresolved safety incident remains; thresholds require local agreement.",
11
+ "stop_conditions": [
12
+ "Pause immediately after a verified privacy leak or a serious unsafe or unauthorized customer commitment; investigate before resuming.",
13
+ "Pause expansion if blinded quality review crosses a pre-agreed noninferiority margin, including within experience strata."
14
+ ],
15
+ "extensions": {
16
+ "gap_id": "G-001",
17
+ "status": "proposed_not_executed"
18
+ }
19
+ }
@@ -0,0 +1,4 @@
1
+ {"evidence_id": "E-001", "source_id": "S-001", "study_id": "ST-001", "sample_id": "SAMPLE-support-all", "claim_id": "C-001", "title": "Generative AI at Work", "year": 2025, "study_type": "quasi_experimental", "population": "Customer-support agents", "sample_size": 5172, "outcome_type": "completion_time", "outcome_measure": "Chat handling time; separate throughput summary: about 15% more issues resolved per hour.", "claim": "AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.", "direction": "support", "relation_to_claim": "support", "effect_direction": "positive", "decision_relation": "conditional", "source_location": "https://academic.oup.com/qje/article/140/2/889/7990658", "limitations": ["One firm, one tool and a staggered nonrandom rollout; causal interpretation depends on identification assumptions."], "status": "SUPPORTED", "extensions": {"domain": "policy", "policy_outcome": "policy_effectiveness", "directness": "direct", "raw_result": {"metric": "issues_resolved_per_hour_relative_change", "value": 15, "unit": "percent", "role": "separate_throughput_measure_not_completion_time", "ci_lower": null, "ci_upper": null, "p_value": null, "uncertainty_status": "not_extracted_for_this_summary_estimand"}, "standardized_effect": null, "study_total_n": 5172, "note": "Row-specific sample; study_total_n must not replace the analyzed sample. E-004 shares the S-001 cohort and is not separately counted."}}
2
+ {"evidence_id": "E-002", "source_id": "S-002", "study_id": "ST-002", "sample_id": "SAMPLE-writing", "claim_id": "C-002", "title": "Experimental evidence on the productivity effects of generative artificial intelligence", "year": 2023, "study_type": "rct", "population": "College-educated working professionals", "sample_size": 453, "outcome_type": "completion_time", "outcome_measure": "Self-reported task time: approximately 40% lower; assessed writing quality approximately 18% higher is a separate measure.", "claim": "ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.", "direction": "support", "relation_to_claim": "support", "effect_direction": "positive", "decision_relation": "conditional", "source_location": "https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf", "limitations": ["Indirect evidence, not directly generalizable: brief incentivized writing tasks did not demand precise factual accuracy or customer-specific context."], "status": "SUPPORTED", "extensions": {"domain": "policy", "policy_outcome": "policy_effectiveness", "directness": "indirect", "raw_result": {"metric": "task_time_relative_change", "value": -40, "unit": "percent", "ci_lower": null, "ci_upper": null, "p_value": null, "uncertainty_status": "not_extracted_for_this_summary_estimand"}, "standardized_effect": null, "study_total_n": 453, "note": "Row-specific sample; study_total_n must not replace the analyzed sample. E-004 shares the S-001 cohort and is not separately counted."}}
3
+ {"evidence_id": "E-003", "source_id": "S-003", "study_id": "ST-003", "sample_id": "SAMPLE-consulting-outside", "claim_id": "C-003", "title": "Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality", "year": 2026, "study_type": "rct", "population": "BCG consultants in the outside-frontier experiment", "sample_size": 373, "outcome_type": "accuracy", "outcome_measure": "Correct business recommendation: about 19 percentage points lower across AI arms; Table 7 has 373 participants, not 758.", "claim": "AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.", "direction": "support", "relation_to_claim": "support", "effect_direction": "negative", "decision_relation": "conditional", "source_location": "https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838", "limitations": ["Indirect evidence, not directly generalizable: consultants solving an experimental business case, not live customer tickets."], "status": "SUPPORTED", "extensions": {"domain": "policy", "policy_outcome": "implementation_risk", "directness": "indirect", "raw_result": {"metric": "correctness_absolute_change", "value": -19, "unit": "percentage_points", "ci_lower": null, "ci_upper": null, "p_value": null, "uncertainty_status": "not_extracted_for_this_summary_estimand"}, "standardized_effect": null, "study_total_n": 758, "note": "Row-specific sample; study_total_n must not replace the analyzed sample. E-004 shares the S-001 cohort and is not separately counted."}}
4
+ {"evidence_id": "E-004", "source_id": "S-001", "study_id": "ST-001", "sample_id": "SAMPLE-support-all", "claim_id": "C-004", "title": "Generative AI at Work", "year": 2025, "study_type": "quasi_experimental", "population": "Experienced and high-skill customer-support agents", "sample_size": null, "outcome_type": "accuracy", "outcome_measure": "Small quality declines among the most experienced and highest-skilled support staff.", "claim": "Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits.", "direction": "support", "relation_to_claim": "support", "effect_direction": "negative", "decision_relation": "conditional", "source_location": "https://academic.oup.com/qje/article/140/2/889/7990658", "limitations": ["Same study as E-001; subgroup sample size was not extracted. This is not an independent replication."], "status": "SUPPORTED", "extensions": {"domain": "policy", "policy_outcome": "implementation_risk", "directness": "direct", "raw_result": {"metric": "experienced_staff_quality", "value": null, "unit": "not_extracted", "ci_lower": null, "ci_upper": null, "p_value": null, "uncertainty_status": "not_extracted_for_this_summary_estimand"}, "standardized_effect": null, "study_total_n": 5172, "note": "Row-specific sample; study_total_n must not replace the analyzed sample. E-004 shares the S-001 cohort and is not separately counted."}}
@@ -0,0 +1,444 @@
1
+ {
2
+ "project_id": "workplace-ai-assistant",
3
+ "revision_id": 1,
4
+ "intent": {
5
+ "domain": "policy",
6
+ "question": "Should an enterprise customer-support team introduce a generative AI assistant?",
7
+ "data_origin": "manual_curated",
8
+ "benchmark_eligible": false,
9
+ "timestamp_note": "Date-only curation marker, not execution time."
10
+ },
11
+ "created_at": "2026-09-08T00:00:00Z",
12
+ "updated_at": "2026-09-08T00:00:00Z",
13
+ "audit_warnings": [],
14
+ "papers": {
15
+ "S-001": {
16
+ "paper_id": "S-001",
17
+ "title": "Generative AI at Work",
18
+ "authors": [
19
+ "Erik Brynjolfsson",
20
+ "Danielle Li",
21
+ "Lindsey Raymond"
22
+ ],
23
+ "year": 2025,
24
+ "venue": "Academic Publication",
25
+ "doi": "10.1093/qje/qjae044",
26
+ "url": "https://academic.oup.com/qje/article/140/2/889/7990658",
27
+ "authority_tier": 1,
28
+ "peer_reviewed": true,
29
+ "summary": ""
30
+ },
31
+ "S-002": {
32
+ "paper_id": "S-002",
33
+ "title": "Experimental evidence on the productivity effects of generative artificial intelligence",
34
+ "authors": [
35
+ "Shakked Noy",
36
+ "Whitney Zhang"
37
+ ],
38
+ "year": 2023,
39
+ "venue": "Academic Publication",
40
+ "doi": "10.1126/science.adh2586",
41
+ "url": "https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf",
42
+ "authority_tier": 1,
43
+ "peer_reviewed": true,
44
+ "summary": ""
45
+ },
46
+ "S-003": {
47
+ "paper_id": "S-003",
48
+ "title": "Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality",
49
+ "authors": [
50
+ "Fabrizio Dell’Acqua",
51
+ "Edward McFowland III",
52
+ "Ethan Mollick",
53
+ "Hila Lifshitz-Assaf",
54
+ "Katherine C. Kellogg",
55
+ "Saran Rajendran",
56
+ "Lisa Krayer",
57
+ "François Candelon",
58
+ "Karim R. Lakhani"
59
+ ],
60
+ "year": 2026,
61
+ "venue": "Academic Publication",
62
+ "doi": "10.1287/orsc.2025.21838",
63
+ "url": "https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838",
64
+ "authority_tier": 1,
65
+ "peer_reviewed": true,
66
+ "summary": ""
67
+ }
68
+ },
69
+ "evidence": {
70
+ "E-001": {
71
+ "evidence_id": "E-001",
72
+ "paper_id": "S-001",
73
+ "outcome_metric": "Chat handling time; separate throughput summary: about 15% more issues resolved per hour.",
74
+ "outcome_dimension": "Policy",
75
+ "claim_id": "C-001",
76
+ "outcome_id": "completion_time",
77
+ "effect_size": {
78
+ "metric": "not_standardized",
79
+ "value": null,
80
+ "ci_lower": null,
81
+ "ci_upper": null,
82
+ "p_value": null
83
+ },
84
+ "sample_size": 5172,
85
+ "sample_description": "Customer-support agents",
86
+ "study_design": "quasi_experimental",
87
+ "direction": "SUPPORTS",
88
+ "confidence_score": null,
89
+ "wwc_rating": "Not applicable: organizational policy",
90
+ "key_quote": "",
91
+ "calibrated_weight": null,
92
+ "bias_flag": "One firm, one tool and a staggered nonrandom rollout; causal interpretation depends on identification assumptions."
93
+ },
94
+ "E-002": {
95
+ "evidence_id": "E-002",
96
+ "paper_id": "S-002",
97
+ "outcome_metric": "Self-reported task time: approximately 40% lower; assessed writing quality approximately 18% higher is a separate measure.",
98
+ "outcome_dimension": "Policy",
99
+ "claim_id": "C-002",
100
+ "outcome_id": "completion_time",
101
+ "effect_size": {
102
+ "metric": "not_standardized",
103
+ "value": null,
104
+ "ci_lower": null,
105
+ "ci_upper": null,
106
+ "p_value": null
107
+ },
108
+ "sample_size": 453,
109
+ "sample_description": "College-educated working professionals",
110
+ "study_design": "rct",
111
+ "direction": "SUPPORTS",
112
+ "confidence_score": null,
113
+ "wwc_rating": "Not applicable: organizational policy",
114
+ "key_quote": "",
115
+ "calibrated_weight": null,
116
+ "bias_flag": "Indirect evidence, not directly generalizable: brief incentivized writing tasks did not demand precise factual accuracy or customer-specific context."
117
+ },
118
+ "E-003": {
119
+ "evidence_id": "E-003",
120
+ "paper_id": "S-003",
121
+ "outcome_metric": "Correct business recommendation: about 19 percentage points lower across AI arms; Table 7 has 373 participants, not 758.",
122
+ "outcome_dimension": "Policy",
123
+ "claim_id": "C-003",
124
+ "outcome_id": "accuracy",
125
+ "effect_size": {
126
+ "metric": "not_standardized",
127
+ "value": null,
128
+ "ci_lower": null,
129
+ "ci_upper": null,
130
+ "p_value": null
131
+ },
132
+ "sample_size": 373,
133
+ "sample_description": "BCG consultants in the outside-frontier experiment",
134
+ "study_design": "rct",
135
+ "direction": "SUPPORTS",
136
+ "confidence_score": null,
137
+ "wwc_rating": "Not applicable: organizational policy",
138
+ "key_quote": "",
139
+ "calibrated_weight": null,
140
+ "bias_flag": "Indirect evidence, not directly generalizable: consultants solving an experimental business case, not live customer tickets."
141
+ },
142
+ "E-004": {
143
+ "evidence_id": "E-004",
144
+ "paper_id": "S-001",
145
+ "outcome_metric": "Small quality declines among the most experienced and highest-skilled support staff.",
146
+ "outcome_dimension": "Policy",
147
+ "claim_id": "C-004",
148
+ "outcome_id": "accuracy",
149
+ "effect_size": {
150
+ "metric": "not_standardized",
151
+ "value": null,
152
+ "ci_lower": null,
153
+ "ci_upper": null,
154
+ "p_value": null
155
+ },
156
+ "sample_size": null,
157
+ "sample_description": "Experienced and high-skill customer-support agents",
158
+ "study_design": "quasi_experimental",
159
+ "direction": "SUPPORTS",
160
+ "confidence_score": null,
161
+ "wwc_rating": "Not applicable: organizational policy",
162
+ "key_quote": "",
163
+ "calibrated_weight": null,
164
+ "bias_flag": "Same study as E-001; subgroup sample size was not extracted. This is not an independent replication."
165
+ }
166
+ },
167
+ "outcomes": {
168
+ "completion_time": {
169
+ "outcome_id": "completion_time",
170
+ "name": "completion_time",
171
+ "dimension": "Policy",
172
+ "category": "Policy",
173
+ "description": ""
174
+ },
175
+ "accuracy": {
176
+ "outcome_id": "accuracy",
177
+ "name": "accuracy",
178
+ "dimension": "Policy",
179
+ "category": "Policy",
180
+ "description": ""
181
+ }
182
+ },
183
+ "claims": {
184
+ "C-001": {
185
+ "claim_id": "C-001",
186
+ "statement": "AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.",
187
+ "outcome_dimension": "Policy",
188
+ "outcome_metric": "",
189
+ "status": "SUPPORTED",
190
+ "pooled_effect_g": null,
191
+ "evidence_ids": [
192
+ "E-001"
193
+ ],
194
+ "bias_warning": ""
195
+ },
196
+ "C-002": {
197
+ "claim_id": "C-002",
198
+ "statement": "ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.",
199
+ "outcome_dimension": "Policy",
200
+ "outcome_metric": "",
201
+ "status": "SUPPORTED",
202
+ "pooled_effect_g": null,
203
+ "evidence_ids": [
204
+ "E-002"
205
+ ],
206
+ "bias_warning": ""
207
+ },
208
+ "C-003": {
209
+ "claim_id": "C-003",
210
+ "statement": "AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.",
211
+ "outcome_dimension": "Policy",
212
+ "outcome_metric": "",
213
+ "status": "SUPPORTED",
214
+ "pooled_effect_g": null,
215
+ "evidence_ids": [
216
+ "E-003"
217
+ ],
218
+ "bias_warning": ""
219
+ },
220
+ "C-004": {
221
+ "claim_id": "C-004",
222
+ "statement": "Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits.",
223
+ "outcome_dimension": "Policy",
224
+ "outcome_metric": "",
225
+ "status": "SUPPORTED",
226
+ "pooled_effect_g": null,
227
+ "evidence_ids": [
228
+ "E-004"
229
+ ],
230
+ "bias_warning": ""
231
+ }
232
+ },
233
+ "risks": {},
234
+ "gaps": {
235
+ "G-001": {
236
+ "gap_id": "G-001",
237
+ "gap_type": "Local applicability and safety",
238
+ "description": "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set.",
239
+ "target_outcome": "",
240
+ "existing_evidence_summary": "E-001 through E-004",
241
+ "recommended_trial_design": "Preregister an intention-to-treat comparison with team-clustered uncertainty and subgroup estimates. Set sample size and noninferiority margins from local baseline variance and operational priorities before enrollment."
242
+ }
243
+ },
244
+ "decision": {
245
+ "decision_id": "D-001",
246
+ "verdict": "PILOT",
247
+ "confidence_score": null,
248
+ "rationale": "A supervised pilot is warranted because one direct field study supports efficiency gains but also reveals heterogeneous quality effects. Indirect experiments identify task boundaries, while local safety and net value remain unknown.",
249
+ "applicability_boundary": "",
250
+ "intervention_plan": {
251
+ "decision": "pilot",
252
+ "target_population": "Enterprise customer-support staff, stratified by tenure and baseline skill.",
253
+ "pilot_duration": "Proposed: two baseline weeks and six pilot weeks; not executed.",
254
+ "ai_usage_policy": "Start with low-risk, knowledge-base-covered queues; agents inspect and edit every proposed reply.",
255
+ "risk_control": [
256
+ "Start with low-risk, knowledge-base-covered queues; agents inspect and edit every proposed reply.",
257
+ "Stratify by tenure and baseline skill; retain expert discretion and measure review workload.",
258
+ "Minimize and redact customer data, restrict access and retention, and review supplier data-use terms before a pilot.",
259
+ "Escalate unfamiliar, ambiguous, sensitive or high-stakes requests to qualified humans; no autonomous commitments."
260
+ ],
261
+ "stop_conditions": [
262
+ "Pause immediately after a verified privacy leak or a serious unsafe or unauthorized customer commitment; investigate before resuming.",
263
+ "Pause expansion if blinded quality review crosses a pre-agreed noninferiority margin, including within experience strata."
264
+ ],
265
+ "evidence_alignment": [
266
+ "E-001",
267
+ "E-002",
268
+ "E-003",
269
+ "E-004"
270
+ ],
271
+ "extensions": {
272
+ "gap_id": "G-001",
273
+ "target_population": "enterprise_customer_support_staff",
274
+ "status": "proposed_not_executed"
275
+ }
276
+ },
277
+ "evaluation_plan": {
278
+ "research_question": "Should an enterprise customer-support team introduce a generative AI assistant?",
279
+ "groups": {
280
+ "treatment": "Eligible teams randomly assigned to supervised assistant access, stratified by tenure and baseline performance.",
281
+ "comparison": "Concurrent teams retaining the existing workflow; document contamination."
282
+ },
283
+ "baseline": "Record resolution rate, paid hours, repeat contacts, blinded quality and review costs before allocation.",
284
+ "post_test": "Assess the same outcomes at pilot end; separately audit privacy and unsafe commitments.",
285
+ "analysis_plan": "Preregister an intention-to-treat comparison with team-clustered uncertainty and subgroup estimates. Set sample size and noninferiority margins from local baseline variance and operational priorities before enrollment.",
286
+ "success_threshold": "Expand only if quality is noninferior, verified resolutions per paid hour improve, net cost is acceptable and no serious unresolved safety incident remains; thresholds require local agreement.",
287
+ "stop_conditions": [
288
+ "Pause immediately after a verified privacy leak or a serious unsafe or unauthorized customer commitment; investigate before resuming.",
289
+ "Pause expansion if blinded quality review crosses a pre-agreed noninferiority margin, including within experience strata."
290
+ ],
291
+ "extensions": {
292
+ "gap_id": "G-001",
293
+ "status": "proposed_not_executed"
294
+ }
295
+ },
296
+ "stop_conditions": [
297
+ "Pause immediately after a verified privacy leak or a serious unsafe or unauthorized customer commitment; investigate before resuming.",
298
+ "Pause expansion if blinded quality review crosses a pre-agreed noninferiority margin, including within experience strata."
299
+ ],
300
+ "timestamp": "2026-09-08T00:00:00Z"
301
+ },
302
+ "edges": [
303
+ {
304
+ "source_id": "E-001",
305
+ "target_id": "S-001",
306
+ "relation": "EXTRACTED_FROM",
307
+ "weight": 1.0,
308
+ "metadata": {}
309
+ },
310
+ {
311
+ "source_id": "E-001",
312
+ "target_id": "C-001",
313
+ "relation": "SUPPORTS",
314
+ "weight": 1.0,
315
+ "metadata": {}
316
+ },
317
+ {
318
+ "source_id": "E-001",
319
+ "target_id": "completion_time",
320
+ "relation": "MEASURES",
321
+ "weight": 1.0,
322
+ "metadata": {}
323
+ },
324
+ {
325
+ "source_id": "E-001",
326
+ "target_id": "G-001",
327
+ "relation": "IDENTIFIES_GAP",
328
+ "weight": 1.0,
329
+ "metadata": {}
330
+ },
331
+ {
332
+ "source_id": "C-001",
333
+ "target_id": "D-001",
334
+ "relation": "GROUNDS_DECISION",
335
+ "weight": 1.0,
336
+ "metadata": {}
337
+ },
338
+ {
339
+ "source_id": "E-002",
340
+ "target_id": "S-002",
341
+ "relation": "EXTRACTED_FROM",
342
+ "weight": 1.0,
343
+ "metadata": {}
344
+ },
345
+ {
346
+ "source_id": "E-002",
347
+ "target_id": "C-002",
348
+ "relation": "SUPPORTS",
349
+ "weight": 1.0,
350
+ "metadata": {}
351
+ },
352
+ {
353
+ "source_id": "E-002",
354
+ "target_id": "completion_time",
355
+ "relation": "MEASURES",
356
+ "weight": 1.0,
357
+ "metadata": {}
358
+ },
359
+ {
360
+ "source_id": "E-002",
361
+ "target_id": "G-001",
362
+ "relation": "IDENTIFIES_GAP",
363
+ "weight": 1.0,
364
+ "metadata": {}
365
+ },
366
+ {
367
+ "source_id": "C-002",
368
+ "target_id": "D-001",
369
+ "relation": "GROUNDS_DECISION",
370
+ "weight": 1.0,
371
+ "metadata": {}
372
+ },
373
+ {
374
+ "source_id": "E-003",
375
+ "target_id": "S-003",
376
+ "relation": "EXTRACTED_FROM",
377
+ "weight": 1.0,
378
+ "metadata": {}
379
+ },
380
+ {
381
+ "source_id": "E-003",
382
+ "target_id": "C-003",
383
+ "relation": "SUPPORTS",
384
+ "weight": 1.0,
385
+ "metadata": {}
386
+ },
387
+ {
388
+ "source_id": "E-003",
389
+ "target_id": "accuracy",
390
+ "relation": "MEASURES",
391
+ "weight": 1.0,
392
+ "metadata": {}
393
+ },
394
+ {
395
+ "source_id": "E-003",
396
+ "target_id": "G-001",
397
+ "relation": "IDENTIFIES_GAP",
398
+ "weight": 1.0,
399
+ "metadata": {}
400
+ },
401
+ {
402
+ "source_id": "C-003",
403
+ "target_id": "D-001",
404
+ "relation": "GROUNDS_DECISION",
405
+ "weight": 1.0,
406
+ "metadata": {}
407
+ },
408
+ {
409
+ "source_id": "E-004",
410
+ "target_id": "S-001",
411
+ "relation": "EXTRACTED_FROM",
412
+ "weight": 1.0,
413
+ "metadata": {}
414
+ },
415
+ {
416
+ "source_id": "E-004",
417
+ "target_id": "C-004",
418
+ "relation": "SUPPORTS",
419
+ "weight": 1.0,
420
+ "metadata": {}
421
+ },
422
+ {
423
+ "source_id": "E-004",
424
+ "target_id": "accuracy",
425
+ "relation": "MEASURES",
426
+ "weight": 1.0,
427
+ "metadata": {}
428
+ },
429
+ {
430
+ "source_id": "E-004",
431
+ "target_id": "G-001",
432
+ "relation": "IDENTIFIES_GAP",
433
+ "weight": 1.0,
434
+ "metadata": {}
435
+ },
436
+ {
437
+ "source_id": "C-004",
438
+ "target_id": "D-001",
439
+ "relation": "GROUNDS_DECISION",
440
+ "weight": 1.0,
441
+ "metadata": {}
442
+ }
443
+ ]
444
+ }
@@ -0,0 +1,41 @@
1
+ {
2
+ "question": "Should an enterprise customer-support team introduce a generative AI assistant?",
3
+ "decision_object": "adopt",
4
+ "intervention": {
5
+ "policy_name": "human_supervised_customer_support_assistant",
6
+ "policy_type": "institutional_reform",
7
+ "mechanism": "approved_knowledge_base_and_agent_review"
8
+ },
9
+ "population": {
10
+ "target_group": "enterprise_customer_support_staff",
11
+ "excluded_groups": "autonomous_agents_and_high_stakes_specialist_advice"
12
+ },
13
+ "comparison": "Existing support workflow without generative AI suggestions.",
14
+ "outcomes": {
15
+ "primary": [
16
+ "policy_effectiveness",
17
+ "implementation_risk"
18
+ ],
19
+ "secondary": [
20
+ "cost_effectiveness",
21
+ "equity",
22
+ "feasibility"
23
+ ]
24
+ },
25
+ "context": {
26
+ "policy_environment": "enterprise_customer_support"
27
+ },
28
+ "scope": {
29
+ "time_range": "2023–2026",
30
+ "evidence_types": [
31
+ "quasi_experimental",
32
+ "rct"
33
+ ]
34
+ },
35
+ "success_condition": "Improve verified resolutions per paid staff hour while preserving service quality, privacy and staff autonomy.",
36
+ "extensions": {
37
+ "domain": "policy",
38
+ "data_origin": "manual_curated",
39
+ "note": "Purposive evidence selection, not a systematic review or model execution benchmark."
40
+ }
41
+ }