eduevidence 5.2.0 → 6.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (386) hide show
  1. package/CONTRIBUTING.md +105 -0
  2. package/README.md +142 -75
  3. package/README.zh-CN.md +73 -30
  4. package/SKILL.md +397 -131
  5. package/agents/openai.yaml +4 -0
  6. package/assets/readme/controlled-execution.svg +34 -0
  7. package/assets/readme/landing-tour.gif +0 -0
  8. package/assets/readme/logo.png +0 -0
  9. package/assets/readme/research-workflow.svg +56 -0
  10. package/assets/readme/studio-graph.png +0 -0
  11. package/assets/readme/studio-overview.png +0 -0
  12. package/assets/readme/studio-reports.png +0 -0
  13. package/assets/readme/studio-tour.gif +0 -0
  14. package/autoevolve/config.yaml +17 -0
  15. package/autoevolve/program.md +25 -0
  16. package/autoevolve/protected.manifest.yaml +34 -0
  17. package/benchmarks/adversarial/cases.jsonl +7 -0
  18. package/benchmarks/evidence-library.json +5268 -0
  19. package/benchmarks/partitions.json +8 -0
  20. package/bin/eduevidence.js +2 -1
  21. package/docs/architecture.md +496 -0
  22. package/docs/autoresearch-evolution-plan.md +2903 -0
  23. package/docs/autoresearch-implementation-status.md +101 -0
  24. package/docs/demo-storyboard.md +20 -0
  25. package/docs/demo-workplace-ai.md +92 -0
  26. package/docs/demo.md +32 -0
  27. package/docs/install-guide.md +150 -0
  28. package/docs/orchestration-role-model.md +1254 -0
  29. package/docs/release-closeout/README.md +17 -0
  30. package/docs/release-closeout/frontend-acceptance.md +23 -0
  31. package/docs/release-closeout/issues.md +19 -0
  32. package/docs/release-closeout/verification.md +28 -0
  33. package/docs/release-contract.md +108 -0
  34. package/docs/research-studio-guide.zh-CN.md +166 -0
  35. package/docs/sciverse-api.md +125 -0
  36. package/eduevidence_cli.py +29 -13
  37. package/engine/_resources.py +13 -0
  38. package/engine/autoevolve/__init__.py +3 -0
  39. package/engine/autoevolve/agent_view.py +167 -0
  40. package/engine/autoevolve/core.py +357 -0
  41. package/engine/autoevolve/events.py +11 -0
  42. package/engine/autoevolve/git_workspace.py +77 -0
  43. package/engine/autoevolve/projection.py +23 -0
  44. package/engine/autoevolve/runner.py +413 -0
  45. package/engine/autoevolve/trust.py +146 -0
  46. package/engine/autoresearch/__init__.py +6 -0
  47. package/engine/autoresearch/commit.py +132 -0
  48. package/engine/autoresearch/contracts.py +126 -0
  49. package/engine/autoresearch/controller.py +207 -0
  50. package/engine/autoresearch/events.py +12 -0
  51. package/engine/autoresearch/gap_priority.py +168 -0
  52. package/engine/autoresearch/projection.py +30 -0
  53. package/engine/autoresearch/research_memory.py +59 -0
  54. package/engine/autoresearch/saturation.py +91 -0
  55. package/engine/briefs.py +2 -1
  56. package/engine/capabilities.py +1 -0
  57. package/engine/contracts.py +3 -1
  58. package/engine/decision_policy.py +96 -0
  59. package/engine/evidence_graph.py +14 -10
  60. package/engine/evidencecore.py +7 -5
  61. package/engine/gaps.py +132 -73
  62. package/engine/ids.py +2 -0
  63. package/engine/judge_pack.py +65 -0
  64. package/engine/library.py +6 -2
  65. package/engine/library_builtin.py +3 -1
  66. package/engine/living.py +36 -5
  67. package/engine/meta_synthesis.py +3 -1
  68. package/engine/migration.py +88 -3
  69. package/engine/orchestration.py +460 -0
  70. package/engine/paths.py +2 -0
  71. package/engine/pilot.py +36 -33
  72. package/engine/project.py +2 -2
  73. package/engine/research_service.py +113 -0
  74. package/engine/studio_read_model.py +400 -0
  75. package/engine/taxonomy.py +211 -0
  76. package/engine/tribunal.py +44 -33
  77. package/engine/update.py +1 -0
  78. package/engine/versions.py +1 -1
  79. package/engine/worker_result.py +109 -0
  80. package/engine/workflows.py +70 -0
  81. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +2934 -0
  82. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
  83. package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
  84. package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
  85. package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
  86. package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
  87. package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
  88. package/examples/ai-coding-assistant-evidence/frame.json +48 -0
  89. package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
  90. package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
  91. package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
  92. package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
  93. package/examples/ai-coding-assistant-evidence/report_spec.json +230 -0
  94. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2934 -0
  95. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2934 -0
  96. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2934 -0
  97. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2934 -0
  98. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2934 -0
  99. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +2934 -0
  100. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +2934 -0
  101. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +2934 -0
  102. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +2934 -0
  103. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +2934 -0
  104. package/examples/ai-coding-assistant-evidence/result.json +1457 -0
  105. package/examples/ai-coding-assistant-evidence/result.zh.json +1457 -0
  106. package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
  107. package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
  108. package/examples/ai-coding-assistant-evidence/verdict.json +107 -0
  109. package/examples/spaced-retrieval-practice/applicability.json +14 -0
  110. package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
  111. package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
  112. package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
  113. package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
  114. package/examples/spaced-retrieval-practice/frame.json +58 -0
  115. package/examples/spaced-retrieval-practice/gate_report.json +101 -0
  116. package/examples/spaced-retrieval-practice/methodology.json +78 -0
  117. package/examples/spaced-retrieval-practice/report_spec.json +212 -0
  118. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
  119. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
  120. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
  121. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
  122. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
  123. package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
  124. package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
  125. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
  126. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
  127. package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
  128. package/examples/spaced-retrieval-practice/result.json +942 -0
  129. package/examples/spaced-retrieval-practice/result.zh.json +942 -0
  130. package/examples/spaced-retrieval-practice/skeptic.json +70 -0
  131. package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
  132. package/examples/spaced-retrieval-practice/verdict.json +93 -0
  133. package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
  134. package/examples/workplace-ai-assistant/claims.jsonl +4 -0
  135. package/examples/workplace-ai-assistant/evaluation.json +19 -0
  136. package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
  137. package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
  138. package/examples/workplace-ai-assistant/final_verdict.json +78 -0
  139. package/examples/workplace-ai-assistant/frame.json +41 -0
  140. package/examples/workplace-ai-assistant/gate_report.json +101 -0
  141. package/examples/workplace-ai-assistant/intervention.json +27 -0
  142. package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
  143. package/examples/workplace-ai-assistant/methodology.json +60 -0
  144. package/examples/workplace-ai-assistant/report_spec.json +224 -0
  145. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2814 -0
  146. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2814 -0
  147. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2814 -0
  148. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2814 -0
  149. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2814 -0
  150. package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
  151. package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
  152. package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
  153. package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
  154. package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
  155. package/examples/workplace-ai-assistant/result.json +615 -0
  156. package/examples/workplace-ai-assistant/result.zh.json +615 -0
  157. package/examples/workplace-ai-assistant/search_log.json +19 -0
  158. package/examples/workplace-ai-assistant/skeptic.json +72 -0
  159. package/examples/workplace-ai-assistant/sources.jsonl +3 -0
  160. package/examples/workplace-ai-assistant/validation_result.json +9 -0
  161. package/examples/workplace-ai-assistant/verdict.json +78 -0
  162. package/install.sh +7 -7
  163. package/integrations/agent_mcp.py +2 -2
  164. package/integrations/orchestration_dispatch.py +146 -0
  165. package/package.json +46 -3
  166. package/pyproject.toml +14 -22
  167. package/references/autoresearch.md +30 -0
  168. package/references/evaluation-policy.md +24 -0
  169. package/references/orchestration.md +22 -0
  170. package/references/report-copy-style.md +67 -0
  171. package/references/retrieval-compliance.md +75 -0
  172. package/references/retrieval-protocol.md +20 -0
  173. package/references/scientific-invariants.md +19 -0
  174. package/retrieval/audit.py +178 -0
  175. package/retrieval/fetch.py +96 -0
  176. package/retrieval/sciverse.py +398 -0
  177. package/retrieval/search.py +47 -7
  178. package/schemas/applicability.schema.json +94 -0
  179. package/schemas/chart-spec.schema.json +10 -3
  180. package/schemas/evidence.schema.json +316 -43
  181. package/schemas/fetch-result.schema.json +2 -1
  182. package/schemas/intervention.schema.json +106 -21
  183. package/schemas/report-result.schema.json +12 -4
  184. package/schemas/report-spec.schema.json +98 -100
  185. package/schemas/skeptic.schema.json +86 -0
  186. package/schemas/source.schema.json +21 -2
  187. package/schemas/v2/finding.schema.json +5 -1
  188. package/schemas/v2/methodology-audit.schema.json +5 -1
  189. package/schemas/v2/outcome.schema.json +28 -5
  190. package/schemas/v2/project.schema.json +2 -2
  191. package/schemas/v2/run.schema.json +1 -1
  192. package/schemas/v2/study.schema.json +5 -1
  193. package/schemas/vNext/autoevolve-session.schema.json +34 -0
  194. package/schemas/vNext/eval-snapshot.schema.json +77 -0
  195. package/schemas/vNext/execution-plan.schema.json +50 -0
  196. package/schemas/vNext/gap-priority.schema.json +54 -0
  197. package/schemas/vNext/negative-search-record.schema.json +68 -0
  198. package/schemas/vNext/research-iteration.schema.json +87 -0
  199. package/schemas/vNext/research-strategy.schema.json +62 -0
  200. package/schemas/vNext/skill-experiment.schema.json +90 -0
  201. package/schemas/vNext/task-spec.schema.json +156 -0
  202. package/schemas/vNext/worker-result.schema.json +60 -0
  203. package/schemas/verdict.schema.json +164 -28
  204. package/scripts/benchmark_judge.py +2 -2
  205. package/scripts/benchmark_v3.py +26 -43
  206. package/scripts/build_esl_artifacts.py +4 -4
  207. package/scripts/build_evidence_library.py +2 -2
  208. package/scripts/build_gh_pages.py +98 -0
  209. package/scripts/build_readme_diagrams.py +72 -0
  210. package/scripts/build_report_variants.py +101 -0
  211. package/scripts/build_result.py +74 -9
  212. package/scripts/check_autoresearch_invariants.py +95 -0
  213. package/scripts/check_package_parity.py +85 -0
  214. package/scripts/check_protocol_alignment.py +375 -0
  215. package/scripts/check_versioned_schemas.py +254 -0
  216. package/scripts/claim_audit.py +13 -8
  217. package/scripts/compute_confidence.py +10 -0
  218. package/scripts/daily_evolve.py +30 -0
  219. package/scripts/dashboard_server.py +130 -101
  220. package/scripts/did_regression.py +17 -32
  221. package/scripts/enrich_projects_human_and_lieflat.py +1 -1
  222. package/scripts/evidence_score.py +5 -2
  223. package/scripts/generate_metrics.py +4 -3
  224. package/scripts/generate_new_projects.py +5 -5
  225. package/scripts/orchestrator.py +286 -36
  226. package/scripts/pre_verdict_gate.py +224 -26
  227. package/scripts/quickstart.py +18 -2
  228. package/scripts/rebake_all_5themes.py +1 -2
  229. package/scripts/research_auto_cli.py +475 -0
  230. package/scripts/run_workspace.py +24 -8
  231. package/scripts/search_provenance.py +64 -0
  232. package/scripts/serve_web.py +9 -10
  233. package/scripts/skill_lint.py +1 -1
  234. package/scripts/skill_payload.py +81 -0
  235. package/scripts/test_adversarial_empirical.py +26 -19
  236. package/scripts/validate_schema.py +46 -2
  237. package/scripts/vnext_cli.py +133 -0
  238. package/setup.py +12 -0
  239. package/skill/agents/evaluation-designer.md +20 -4
  240. package/skill/agents/evidence-analyst.md +19 -3
  241. package/skill/agents/evidence-judge.md +50 -2
  242. package/skill/agents/evidence-retriever.md +20 -3
  243. package/skill/agents/intervention-designer.md +20 -4
  244. package/skill/agents/method-reviewer.md +18 -2
  245. package/skill/agents/{education-planner.md → research-planner.md} +19 -3
  246. package/skill/agents/skeptic.md +18 -2
  247. package/skill/roles/registry.yaml +45 -0
  248. package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
  249. package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
  250. package/skill/sub-skills/data-analysis/SKILL.md +34 -15
  251. package/skill/sub-skills/ethics-review/SKILL.md +33 -10
  252. package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
  253. package/skill/sub-skills/evidence-review/SKILL.md +31 -12
  254. package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
  255. package/skill/sub-skills/literature-review/SKILL.md +35 -14
  256. package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
  257. package/skill/sub-skills/report-generation/SKILL.md +40 -6
  258. package/skill/sub-skills/research-planning/SKILL.md +41 -14
  259. package/skill/sub-skills/study-design/SKILL.md +30 -9
  260. package/skill/task-briefs/adjudicate.md +32 -7
  261. package/skill/task-briefs/applicability.md +38 -0
  262. package/skill/task-briefs/audit.md +32 -7
  263. package/skill/task-briefs/challenge.md +34 -5
  264. package/skill/task-briefs/evaluate.md +30 -5
  265. package/skill/task-briefs/extract.md +31 -8
  266. package/skill/task-briefs/frame.md +39 -10
  267. package/skill/task-briefs/intervene.md +32 -6
  268. package/skill/task-briefs/present.md +32 -8
  269. package/skill/task-briefs/projection.md +37 -0
  270. package/skill/task-briefs/retrieve.md +36 -6
  271. package/skill/workflows/decision-and-pilot.md +85 -0
  272. package/skill/workflows/evaluate-and-update.md +93 -0
  273. package/skill/workflows/evidence-review.md +117 -0
  274. package/visualization/eduevidence-report/assets/base.css +2 -2
  275. package/visualization/eduevidence-report/assets/reader.css +752 -0
  276. package/visualization/eduevidence-report/assets/reader.js +132 -0
  277. package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
  278. package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
  279. package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
  280. package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
  281. package/visualization/eduevidence-report/scripts/build_report.py +561 -121
  282. package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
  283. package/visualization/eduevidence-report/scripts/lieflat_engine.py +371 -136
  284. package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
  285. package/visualization/eduevidence-report/themes/academic.css +1 -1
  286. package/visualization/eduevidence-report/themes/claude.css +1 -1
  287. package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
  288. package/visualization/eduevidence-report/themes/datalab.css +2 -2
  289. package/visualization/eduevidence-report/themes/presentation.css +2 -2
  290. package/web/README.md +18 -0
  291. package/web/architecture.html +14885 -0
  292. package/web/index.html +53 -0
  293. package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
  294. package/web/studio/assets/index-B8tkF44Q.css +1 -0
  295. package/web/studio/assets/index-CQ6Keoyc.js +230 -0
  296. package/web/studio/config.json +1 -0
  297. package/web/studio/index.html +14 -0
  298. package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
  299. package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
  300. package/engine/__pycache__/bias.cpython-312.pyc +0 -0
  301. package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
  302. package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
  303. package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
  304. package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
  305. package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
  306. package/engine/__pycache__/events.cpython-312.pyc +0 -0
  307. package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
  308. package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
  309. package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
  310. package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
  311. package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
  312. package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
  313. package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
  314. package/engine/__pycache__/ids.cpython-312.pyc +0 -0
  315. package/engine/__pycache__/library.cpython-312.pyc +0 -0
  316. package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
  317. package/engine/__pycache__/living.cpython-312.pyc +0 -0
  318. package/engine/__pycache__/log.cpython-312.pyc +0 -0
  319. package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
  320. package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
  321. package/engine/__pycache__/migration.cpython-312.pyc +0 -0
  322. package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
  323. package/engine/__pycache__/paths.cpython-312.pyc +0 -0
  324. package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
  325. package/engine/__pycache__/planner.cpython-312.pyc +0 -0
  326. package/engine/__pycache__/project.cpython-312.pyc +0 -0
  327. package/engine/__pycache__/projections.cpython-312.pyc +0 -0
  328. package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
  329. package/engine/__pycache__/run.cpython-312.pyc +0 -0
  330. package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
  331. package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
  332. package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
  333. package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
  334. package/engine/__pycache__/update.cpython-312.pyc +0 -0
  335. package/engine/__pycache__/versions.cpython-312.pyc +0 -0
  336. package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
  337. package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
  338. package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
  339. package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
  340. package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
  341. package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
  342. package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
  343. package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
  344. package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
  345. package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
  346. package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
  347. package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
  348. package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
  349. package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
  350. package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
  351. package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
  352. package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
  353. package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
  354. package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
  355. package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
  356. package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
  357. package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
  358. package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
  359. package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
  360. package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
  361. package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
  362. package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
  363. package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
  364. package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
  365. package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
  366. package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
  367. package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
  368. package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
  369. package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
  370. package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
  371. package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
  372. package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
  373. package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
  374. package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
  375. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
  376. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
  377. package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
  378. package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
  379. package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
  380. package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
  381. package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
  382. package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
  383. package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
  384. package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
  385. package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
  386. package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
@@ -0,0 +1,615 @@
1
+ {
2
+ "meta": {
3
+ "skill": "eduevidence",
4
+ "domain": "policy",
5
+ "version": "6.0.0",
6
+ "mode": "platform_native",
7
+ "question": "Should an enterprise customer-support team introduce a generative AI assistant?",
8
+ "data_origin": "manual_curated"
9
+ },
10
+ "research_frame": {
11
+ "question": "Should an enterprise customer-support team introduce a generative AI assistant?",
12
+ "decision_object": "adopt",
13
+ "intervention": {
14
+ "policy_name": "human_supervised_customer_support_assistant",
15
+ "policy_type": "institutional_reform",
16
+ "mechanism": "approved_knowledge_base_and_agent_review"
17
+ },
18
+ "population": {
19
+ "target_group": "enterprise_customer_support_staff",
20
+ "excluded_groups": "autonomous_agents_and_high_stakes_specialist_advice"
21
+ },
22
+ "comparison": "Existing support workflow without generative AI suggestions.",
23
+ "outcomes": {
24
+ "primary": [
25
+ "policy_effectiveness",
26
+ "implementation_risk"
27
+ ],
28
+ "secondary": [
29
+ "cost_effectiveness",
30
+ "equity",
31
+ "feasibility"
32
+ ]
33
+ },
34
+ "context": {
35
+ "policy_environment": "enterprise_customer_support"
36
+ },
37
+ "scope": {
38
+ "time_range": "2023–2026",
39
+ "evidence_types": [
40
+ "quasi_experimental",
41
+ "rct"
42
+ ]
43
+ },
44
+ "success_condition": "Improve verified resolutions per paid staff hour while preserving service quality, privacy and staff autonomy.",
45
+ "extensions": {
46
+ "domain": "policy",
47
+ "data_origin": "manual_curated",
48
+ "note": "Purposive evidence selection, not a systematic review or model execution benchmark."
49
+ }
50
+ },
51
+ "decision": {
52
+ "decision_question": "Should an enterprise customer-support team introduce a generative AI assistant?",
53
+ "target_population": "Enterprise customer-support staff, stratified by tenure and baseline skill.",
54
+ "target_context": "Human-supervised support using an approved knowledge base.",
55
+ "recommended_action": "pilot",
56
+ "confidence": "Moderate",
57
+ "confidence_score": 0.578,
58
+ "confidence_policy_version": "2026-08-12.v3",
59
+ "raw_model_confidence": "Moderate",
60
+ "raw_model_confidence_breakdown": {
61
+ "score": 0.578,
62
+ "evidence_quality": 0.8,
63
+ "consistency": 0.0,
64
+ "directness": 0.75,
65
+ "evidence_count": 4,
66
+ "independent_studies": 3,
67
+ "independent_samples": 3,
68
+ "count_term": 0.75,
69
+ "conflict_penalty": 0.0,
70
+ "unsupported_penalty": 0.0,
71
+ "note": "Adjudicator-stated confidence before the deterministic override."
72
+ },
73
+ "independent_studies": 3,
74
+ "independent_samples": 3,
75
+ "supported_claims": [
76
+ "AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome. — E-001",
77
+ "ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support. — E-002",
78
+ "AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support. — E-003",
79
+ "Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits. — E-004"
80
+ ],
81
+ "uncertain_claims": [
82
+ "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set [no direct evidence]."
83
+ ],
84
+ "decision_rationale": "A supervised pilot is warranted because one direct field study supports efficiency gains but also reveals heterogeneous quality effects. Indirect experiments identify task boundaries, while local safety and net value remain unknown.",
85
+ "strongest_support": "Supervised AI assistance improves handling speed and answer consistency in customer-support work, with quality maintained.",
86
+ "key_uncertainty": "Evidence comes from adjacent writing and advisory settings rather than the support floor, so transfer to live customer conversations is unproven.",
87
+ "main_risk": "Unsupervised or knowledge-base-free use can produce confident wrong answers to customers, and over-reliance erodes agent skill over time.",
88
+ "next_action": "Run a supervised pilot on approved knowledge bases with human review on every reply, and track escalation and correction rates.",
89
+ "methodology_summary": "One staggered-rollout quasi-experiment and two randomized experiments. Only the support study is direct; no pooled standardized effect or model benchmark was computed.",
90
+ "what_can_be_claimed": [
91
+ "Under human supervision on an approved knowledge base, AI assistance can shorten handling time while quality is monitored.",
92
+ "Benefits are not uniform across staff: the most experienced agents need their own quality monitoring."
93
+ ],
94
+ "what_cannot_be_claimed": [
95
+ "Universal gains, autonomous deployment safety, privacy protection, reduced staffing requirements or educational learning gains."
96
+ ],
97
+ "exceeds_evidence_boundary": [
98
+ "Claiming universal gains or that autonomous deployment is safe exceeds the boundary: no included study measures privacy incidents, local net cost or subgroup service quality."
99
+ ],
100
+ "missing_evidence": [
101
+ "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set [no direct evidence]."
102
+ ],
103
+ "applicability": {
104
+ "required_conditions": [
105
+ "Start with low-risk, knowledge-base-covered queues; agents inspect and edit every proposed reply.",
106
+ "Stratify by tenure and baseline skill; retain expert discretion and measure review workload.",
107
+ "Minimize and redact customer data, restrict access and retention, and review supplier data-use terms before a pilot.",
108
+ "Escalate unfamiliar, ambiguous, sensitive or high-stakes requests to qualified humans; no autonomous commitments."
109
+ ]
110
+ },
111
+ "extensions": {
112
+ "data_origin": "manual_curated",
113
+ "benchmark_eligible": false,
114
+ "note": "Confidence and the decision bound come from the deterministic policy in engine/decision_policy.py, enforced by the Pre-Verdict Gate; this record is a curated evidence selection, not a systematic review or a model run.",
115
+ "knowledge_gaps": [
116
+ {
117
+ "gap_id": "G-001",
118
+ "evidence_ids": [
119
+ "E-001",
120
+ "E-002",
121
+ "E-003",
122
+ "E-004"
123
+ ],
124
+ "summary": "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set [no direct evidence]."
125
+ }
126
+ ]
127
+ }
128
+ },
129
+ "sources": [
130
+ {
131
+ "source_id": "S-001",
132
+ "title": "Generative AI at Work",
133
+ "authors": [
134
+ "Erik Brynjolfsson",
135
+ "Danielle Li",
136
+ "Lindsey Raymond"
137
+ ],
138
+ "year": 2025,
139
+ "doi": "10.1093/qje/qjae044",
140
+ "canonical_url": "https://academic.oup.com/qje/article/140/2/889/7990658",
141
+ "source_type": "journal_article",
142
+ "authority_level": "tier1_paper_doi",
143
+ "status": "VALID",
144
+ "source_locator": {
145
+ "type": "section",
146
+ "section": "Abstract; III.B and Table I; IV.A and Table II; IV.B; VIII"
147
+ },
148
+ "fetch": {
149
+ "original_url": "https://academic.oup.com/qje/article/140/2/889/7990658",
150
+ "fetch_method": "native",
151
+ "fetch_provider": "builtin",
152
+ "fetch_status": "FETCH_VALID"
153
+ },
154
+ "extensions": {
155
+ "verified_on": "2026-09-08",
156
+ "data_origin": "manual_curated",
157
+ "version": "published QJE article",
158
+ "verification_scope": "Full primary text read through web tool; no Crossref registry or retraction check claimed."
159
+ }
160
+ },
161
+ {
162
+ "source_id": "S-002",
163
+ "title": "Experimental evidence on the productivity effects of generative artificial intelligence",
164
+ "authors": [
165
+ "Shakked Noy",
166
+ "Whitney Zhang"
167
+ ],
168
+ "year": 2023,
169
+ "doi": "10.1126/science.adh2586",
170
+ "canonical_url": "https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf",
171
+ "source_type": "journal_article",
172
+ "authority_level": "tier1_paper_doi",
173
+ "status": "VALID",
174
+ "source_locator": {
175
+ "type": "section",
176
+ "section": "Author manuscript pp. 1, 3–4, 7; Figure 1 (p. 12)"
177
+ },
178
+ "fetch": {
179
+ "original_url": "https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf",
180
+ "fetch_method": "native",
181
+ "fetch_provider": "builtin",
182
+ "fetch_status": "FETCH_VALID"
183
+ },
184
+ "extensions": {
185
+ "verified_on": "2026-09-08",
186
+ "data_origin": "manual_curated",
187
+ "version": "author manuscript associated with Science 2023; not the March 444-person draft",
188
+ "verification_scope": "Full primary text read through web tool; no Crossref registry or retraction check claimed."
189
+ }
190
+ },
191
+ {
192
+ "source_id": "S-003",
193
+ "title": "Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality",
194
+ "authors": [
195
+ "Fabrizio Dell’Acqua",
196
+ "Edward McFowland III",
197
+ "Ethan Mollick",
198
+ "Hila Lifshitz-Assaf",
199
+ "Katherine C. Kellogg",
200
+ "Saran Rajendran",
201
+ "Lisa Krayer",
202
+ "François Candelon",
203
+ "Karim R. Lakhani"
204
+ ],
205
+ "year": 2026,
206
+ "doi": "10.1287/orsc.2025.21838",
207
+ "canonical_url": "https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838",
208
+ "source_type": "journal_article",
209
+ "authority_level": "tier1_paper_doi",
210
+ "status": "VALID",
211
+ "source_locator": {
212
+ "type": "section",
213
+ "section": "Sections 3 and 4.2; Figure 5; Table 7"
214
+ },
215
+ "fetch": {
216
+ "original_url": "https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838",
217
+ "fetch_method": "native",
218
+ "fetch_provider": "builtin",
219
+ "fetch_status": "FETCH_VALID"
220
+ },
221
+ "extensions": {
222
+ "verified_on": "2026-09-08",
223
+ "data_origin": "manual_curated",
224
+ "version": "published Organization Science 2026 article",
225
+ "verification_scope": "Full primary text read through web tool; no Crossref registry or retraction check claimed."
226
+ }
227
+ }
228
+ ],
229
+ "evidence": [
230
+ {
231
+ "evidence_id": "E-001",
232
+ "source_id": "S-001",
233
+ "study_id": "ST-001",
234
+ "sample_id": "SAMPLE-support-all",
235
+ "claim_id": "C-001",
236
+ "title": "Generative AI at Work",
237
+ "year": 2025,
238
+ "study_type": "quasi_experimental",
239
+ "population": "Customer-support agents",
240
+ "sample_size": 5172,
241
+ "outcome_type": "policy_effectiveness",
242
+ "outcome_measure": "Chat handling time; separate throughput summary: about 15% more issues resolved per hour.",
243
+ "claim": "AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.",
244
+ "direction": "support",
245
+ "relation_to_claim": "support",
246
+ "effect_direction": "positive",
247
+ "decision_relation": "conditional",
248
+ "source_location": "https://academic.oup.com/qje/article/140/2/889/7990658",
249
+ "limitations": [
250
+ "One firm, one tool and a staggered nonrandom rollout; causal interpretation depends on identification assumptions."
251
+ ],
252
+ "status": "SUPPORTED",
253
+ "quality_dimensions": {
254
+ "D1_study_design": 1,
255
+ "D2_sample_quality": 2,
256
+ "D3_measurement_validity": 2,
257
+ "D4_temporal_strength": 2,
258
+ "D5_directness": 2
259
+ },
260
+ "quality_score": 9.0,
261
+ "extensions": {
262
+ "domain": "policy",
263
+ "policy_outcome": "policy_effectiveness",
264
+ "teaching_neutral_outcome_token": "completion_time",
265
+ "directness": "direct",
266
+ "raw_result": {
267
+ "metric": "issues_resolved_per_hour_relative_change",
268
+ "value": 15,
269
+ "unit": "percent",
270
+ "role": "separate_throughput_measure_not_completion_time",
271
+ "ci_lower": null,
272
+ "ci_upper": null,
273
+ "p_value": null,
274
+ "uncertainty_status": "not_extracted_for_this_summary_estimand"
275
+ },
276
+ "standardized_effect": null,
277
+ "study_total_n": 5172,
278
+ "note": "Row-specific sample; study_total_n must not replace the analyzed sample. E-004 shares the S-001 cohort and is not separately counted."
279
+ }
280
+ },
281
+ {
282
+ "evidence_id": "E-002",
283
+ "source_id": "S-002",
284
+ "study_id": "ST-002",
285
+ "sample_id": "SAMPLE-writing",
286
+ "claim_id": "C-002",
287
+ "title": "Experimental evidence on the productivity effects of generative artificial intelligence",
288
+ "year": 2023,
289
+ "study_type": "rct",
290
+ "population": "College-educated working professionals",
291
+ "sample_size": 453,
292
+ "outcome_type": "policy_effectiveness",
293
+ "outcome_measure": "Self-reported task time: approximately 40% lower; assessed writing quality approximately 18% higher is a separate measure.",
294
+ "claim": "ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.",
295
+ "direction": "support",
296
+ "relation_to_claim": "support",
297
+ "effect_direction": "positive",
298
+ "decision_relation": "conditional",
299
+ "source_location": "https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf",
300
+ "limitations": [
301
+ "Indirect evidence, not directly generalizable: brief incentivized writing tasks did not demand precise factual accuracy or customer-specific context."
302
+ ],
303
+ "status": "SUPPORTED",
304
+ "quality_dimensions": {
305
+ "D1_study_design": 2,
306
+ "D2_sample_quality": 2,
307
+ "D3_measurement_validity": 1,
308
+ "D4_temporal_strength": 1,
309
+ "D5_directness": 1
310
+ },
311
+ "quality_score": 7.0,
312
+ "extensions": {
313
+ "domain": "policy",
314
+ "policy_outcome": "policy_effectiveness",
315
+ "teaching_neutral_outcome_token": "completion_time",
316
+ "directness": "indirect",
317
+ "raw_result": {
318
+ "metric": "task_time_relative_change",
319
+ "value": -40,
320
+ "unit": "percent",
321
+ "ci_lower": null,
322
+ "ci_upper": null,
323
+ "p_value": null,
324
+ "uncertainty_status": "not_extracted_for_this_summary_estimand"
325
+ },
326
+ "standardized_effect": null,
327
+ "study_total_n": 453,
328
+ "note": "Row-specific sample; study_total_n must not replace the analyzed sample. E-004 shares the S-001 cohort and is not separately counted."
329
+ }
330
+ },
331
+ {
332
+ "evidence_id": "E-003",
333
+ "source_id": "S-003",
334
+ "study_id": "ST-003",
335
+ "sample_id": "SAMPLE-consulting-outside",
336
+ "claim_id": "C-003",
337
+ "title": "Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality",
338
+ "year": 2026,
339
+ "study_type": "rct",
340
+ "population": "BCG consultants in the outside-frontier experiment",
341
+ "sample_size": 373,
342
+ "outcome_type": "implementation_risk",
343
+ "outcome_measure": "Correct business recommendation: about 19 percentage points lower across AI arms; Table 7 has 373 participants, not 758.",
344
+ "claim": "AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.",
345
+ "direction": "support",
346
+ "relation_to_claim": "support",
347
+ "effect_direction": "negative",
348
+ "decision_relation": "conditional",
349
+ "source_location": "https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838",
350
+ "limitations": [
351
+ "Indirect evidence, not directly generalizable: consultants solving an experimental business case, not live customer tickets."
352
+ ],
353
+ "status": "SUPPORTED",
354
+ "quality_dimensions": {
355
+ "D1_study_design": 2,
356
+ "D2_sample_quality": 2,
357
+ "D3_measurement_validity": 2,
358
+ "D4_temporal_strength": 1,
359
+ "D5_directness": 1
360
+ },
361
+ "quality_score": 8.0,
362
+ "extensions": {
363
+ "domain": "policy",
364
+ "policy_outcome": "implementation_risk",
365
+ "teaching_neutral_outcome_token": "accuracy",
366
+ "directness": "indirect",
367
+ "raw_result": {
368
+ "metric": "correctness_absolute_change",
369
+ "value": -19,
370
+ "unit": "percentage_points",
371
+ "ci_lower": null,
372
+ "ci_upper": null,
373
+ "p_value": null,
374
+ "uncertainty_status": "not_extracted_for_this_summary_estimand"
375
+ },
376
+ "standardized_effect": null,
377
+ "study_total_n": 758,
378
+ "note": "Row-specific sample; study_total_n must not replace the analyzed sample. E-004 shares the S-001 cohort and is not separately counted."
379
+ }
380
+ },
381
+ {
382
+ "evidence_id": "E-004",
383
+ "source_id": "S-001",
384
+ "study_id": "ST-001",
385
+ "sample_id": "SAMPLE-support-all",
386
+ "claim_id": "C-004",
387
+ "title": "Generative AI at Work",
388
+ "year": 2025,
389
+ "study_type": "quasi_experimental",
390
+ "population": "Experienced and high-skill customer-support agents",
391
+ "sample_size": null,
392
+ "outcome_type": "implementation_risk",
393
+ "outcome_measure": "Small quality declines among the most experienced and highest-skilled support staff.",
394
+ "claim": "Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits.",
395
+ "direction": "support",
396
+ "relation_to_claim": "support",
397
+ "effect_direction": "negative",
398
+ "decision_relation": "conditional",
399
+ "source_location": "https://academic.oup.com/qje/article/140/2/889/7990658",
400
+ "limitations": [
401
+ "Same study as E-001; subgroup sample size was not extracted. This is not an independent replication."
402
+ ],
403
+ "status": "SUPPORTED",
404
+ "quality_dimensions": {
405
+ "D1_study_design": 1,
406
+ "D2_sample_quality": 1,
407
+ "D3_measurement_validity": 2,
408
+ "D4_temporal_strength": 2,
409
+ "D5_directness": 2
410
+ },
411
+ "quality_score": 8.0,
412
+ "extensions": {
413
+ "domain": "policy",
414
+ "policy_outcome": "implementation_risk",
415
+ "teaching_neutral_outcome_token": "accuracy",
416
+ "directness": "direct",
417
+ "raw_result": {
418
+ "metric": "experienced_staff_quality",
419
+ "value": null,
420
+ "unit": "not_extracted",
421
+ "ci_lower": null,
422
+ "ci_upper": null,
423
+ "p_value": null,
424
+ "uncertainty_status": "not_extracted_for_this_summary_estimand"
425
+ },
426
+ "standardized_effect": null,
427
+ "study_total_n": 5172,
428
+ "note": "Row-specific sample; study_total_n must not replace the analyzed sample. E-004 shares the S-001 cohort and is not separately counted."
429
+ }
430
+ }
431
+ ],
432
+ "claims": [
433
+ {
434
+ "claim_id": "C-001",
435
+ "claim": "AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.",
436
+ "outcome_type": "policy_effectiveness",
437
+ "evidence_ids": [
438
+ "E-001"
439
+ ],
440
+ "status": "SUPPORTED",
441
+ "pooled_effect_g": null
442
+ },
443
+ {
444
+ "claim_id": "C-002",
445
+ "claim": "ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.",
446
+ "outcome_type": "policy_effectiveness",
447
+ "evidence_ids": [
448
+ "E-002"
449
+ ],
450
+ "status": "SUPPORTED",
451
+ "pooled_effect_g": null
452
+ },
453
+ {
454
+ "claim_id": "C-003",
455
+ "claim": "AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.",
456
+ "outcome_type": "implementation_risk",
457
+ "evidence_ids": [
458
+ "E-003"
459
+ ],
460
+ "status": "SUPPORTED",
461
+ "pooled_effect_g": null
462
+ },
463
+ {
464
+ "claim_id": "C-004",
465
+ "claim": "Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits.",
466
+ "outcome_type": "implementation_risk",
467
+ "evidence_ids": [
468
+ "E-004"
469
+ ],
470
+ "status": "SUPPORTED",
471
+ "pooled_effect_g": null
472
+ }
473
+ ],
474
+ "outcomes": [
475
+ {
476
+ "outcome_type": "policy_effectiveness",
477
+ "positive_count": 2,
478
+ "negative_count": 0,
479
+ "null_count": 0,
480
+ "evidence_ids": [
481
+ "E-001",
482
+ "E-002"
483
+ ]
484
+ },
485
+ {
486
+ "outcome_type": "implementation_risk",
487
+ "positive_count": 0,
488
+ "negative_count": 2,
489
+ "null_count": 0,
490
+ "evidence_ids": [
491
+ "E-003",
492
+ "E-004"
493
+ ]
494
+ }
495
+ ],
496
+ "methodology_reviews": [
497
+ {
498
+ "target": "overall",
499
+ "verdict": "CONCERN",
500
+ "audit_items": {
501
+ "evidence_level": {
502
+ "status": "met"
503
+ },
504
+ "causal_identification": {
505
+ "status": "partial"
506
+ },
507
+ "external_validity": {
508
+ "status": "partial"
509
+ },
510
+ "cost_evidence": {
511
+ "status": "missing"
512
+ },
513
+ "stakeholder_representation": {
514
+ "status": "partial"
515
+ },
516
+ "implementation_evidence": {
517
+ "status": "partial"
518
+ },
519
+ "equity_analysis": {
520
+ "status": "partial"
521
+ },
522
+ "effect_size_reported": {
523
+ "status": "partial"
524
+ },
525
+ "uncertainty_quantified": {
526
+ "status": "partial"
527
+ },
528
+ "comparator_clarity": {
529
+ "status": "met"
530
+ },
531
+ "publication_bias_risk": {
532
+ "status": "partial"
533
+ },
534
+ "generalizability_claims": {
535
+ "status": "met"
536
+ }
537
+ },
538
+ "task_vs_learning_guard": {
539
+ "measured_construct": "workplace_task_performance",
540
+ "equates_task_with_learning": false,
541
+ "note": "Teaching outcomes are not applicable; productivity is not a claim about student learning."
542
+ },
543
+ "limitations": [
544
+ "One firm, one tool and a staggered nonrandom rollout; causal interpretation depends on identification assumptions.",
545
+ "Indirect evidence, not directly generalizable: brief incentivized writing tasks did not demand precise factual accuracy or customer-specific context.",
546
+ "Indirect evidence, not directly generalizable: consultants solving an experimental business case, not live customer tickets.",
547
+ "Same study as E-001; subgroup sample size was not extracted. This is not an independent replication."
548
+ ],
549
+ "suggestions": [
550
+ "Start with low-risk, knowledge-base-covered queues; agents inspect and edit every proposed reply.",
551
+ "Stratify by tenure and baseline skill; retain expert discretion and measure review workload.",
552
+ "Minimize and redact customer data, restrict access and retention, and review supplier data-use terms before a pilot.",
553
+ "Escalate unfamiliar, ambiguous, sensitive or high-stakes requests to qualified humans; no autonomous commitments."
554
+ ],
555
+ "extensions": {}
556
+ }
557
+ ],
558
+ "applicability": {
559
+ "suitable_for": "Start with low-risk, knowledge-base-covered queues; agents inspect and edit every proposed reply.",
560
+ "required_conditions": [
561
+ "Start with low-risk, knowledge-base-covered queues; agents inspect and edit every proposed reply.",
562
+ "Stratify by tenure and baseline skill; retain expert discretion and measure review workload.",
563
+ "Minimize and redact customer data, restrict access and retention, and review supplier data-use terms before a pilot.",
564
+ "Escalate unfamiliar, ambiguous, sensitive or high-stakes requests to qualified humans; no autonomous commitments."
565
+ ],
566
+ "not_suitable_for": "Universal gains, autonomous deployment safety, privacy protection, reduced staffing requirements or educational learning gains."
567
+ },
568
+ "intervention": {
569
+ "decision": "pilot",
570
+ "target_population": "Enterprise customer-support staff, stratified by tenure and baseline skill.",
571
+ "pilot_duration": "Proposed: two baseline weeks and six pilot weeks; not executed.",
572
+ "ai_usage_policy": "Start with low-risk, knowledge-base-covered queues; agents inspect and edit every proposed reply.",
573
+ "risk_control": [
574
+ "Start with low-risk, knowledge-base-covered queues; agents inspect and edit every proposed reply.",
575
+ "Stratify by tenure and baseline skill; retain expert discretion and measure review workload.",
576
+ "Minimize and redact customer data, restrict access and retention, and review supplier data-use terms before a pilot.",
577
+ "Escalate unfamiliar, ambiguous, sensitive or high-stakes requests to qualified humans; no autonomous commitments."
578
+ ],
579
+ "stop_conditions": [
580
+ "Pause immediately after a verified privacy leak or a serious unsafe or unauthorized customer commitment; investigate before resuming.",
581
+ "Pause expansion if blinded quality review crosses a pre-agreed noninferiority margin, including within experience strata."
582
+ ],
583
+ "evidence_alignment": [
584
+ "E-001",
585
+ "E-002",
586
+ "E-003",
587
+ "E-004"
588
+ ],
589
+ "extensions": {
590
+ "gap_id": "G-001",
591
+ "target_population": "enterprise_customer_support_staff",
592
+ "status": "proposed_not_executed"
593
+ }
594
+ },
595
+ "evaluation": {
596
+ "research_question": "Should an enterprise customer-support team introduce a generative AI assistant?",
597
+ "groups": {
598
+ "treatment": "Eligible teams randomly assigned to supervised assistant access, stratified by tenure and baseline performance.",
599
+ "comparison": "Concurrent teams retaining the existing workflow; document contamination."
600
+ },
601
+ "baseline": "Record resolution rate, paid hours, repeat contacts, blinded quality and review costs before allocation.",
602
+ "post_test": "Assess the same outcomes at pilot end; separately audit privacy and unsafe commitments.",
603
+ "analysis_plan": "Preregister an intention-to-treat comparison with team-clustered uncertainty and subgroup estimates. Set sample size and noninferiority margins from local baseline variance and operational priorities before enrollment.",
604
+ "success_threshold": "Expand only if quality is noninferior, verified resolutions per paid hour improve, net cost is acceptable and no serious unresolved safety incident remains; thresholds require local agreement.",
605
+ "stop_conditions": [
606
+ "Pause immediately after a verified privacy leak or a serious unsafe or unauthorized customer commitment; investigate before resuming.",
607
+ "Pause expansion if blinded quality review crosses a pre-agreed noninferiority margin, including within experience strata."
608
+ ],
609
+ "extensions": {
610
+ "gap_id": "G-001",
611
+ "status": "proposed_not_executed"
612
+ }
613
+ },
614
+ "forest_plot_data": []
615
+ }