eduevidence 5.2.0 → 6.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (276) hide show
  1. package/README.md +54 -42
  2. package/README.zh-CN.md +51 -28
  3. package/SKILL.md +390 -133
  4. package/agents/openai.yaml +4 -0
  5. package/assets/readme/controlled-execution.svg +34 -0
  6. package/assets/readme/logo.png +0 -0
  7. package/assets/readme/research-workflow.svg +56 -0
  8. package/assets/readme/studio-graph.png +0 -0
  9. package/assets/readme/studio-overview.png +0 -0
  10. package/assets/readme/studio-reports.png +0 -0
  11. package/autoevolve/config.yaml +17 -0
  12. package/autoevolve/program.md +25 -0
  13. package/autoevolve/protected.manifest.yaml +34 -0
  14. package/benchmarks/adversarial/cases.jsonl +7 -0
  15. package/benchmarks/evidence-library.json +5268 -0
  16. package/benchmarks/partitions.json +8 -0
  17. package/docs/architecture.md +220 -0
  18. package/docs/autoresearch-evolution-plan.md +2903 -0
  19. package/docs/autoresearch-implementation-status.md +101 -0
  20. package/docs/demo-storyboard.md +20 -0
  21. package/docs/demo-workplace-ai.md +92 -0
  22. package/docs/demo.md +32 -0
  23. package/docs/install-guide.md +150 -0
  24. package/docs/orchestration-role-model.md +1254 -0
  25. package/docs/release-closeout/README.md +17 -0
  26. package/docs/release-closeout/frontend-acceptance.md +23 -0
  27. package/docs/release-closeout/issues.md +19 -0
  28. package/docs/release-closeout/verification.md +28 -0
  29. package/docs/release-contract.md +108 -0
  30. package/docs/research-studio-guide.zh-CN.md +166 -0
  31. package/eduevidence_cli.py +17 -11
  32. package/engine/_resources.py +13 -0
  33. package/engine/autoevolve/__init__.py +3 -0
  34. package/engine/autoevolve/agent_view.py +167 -0
  35. package/engine/autoevolve/core.py +357 -0
  36. package/engine/autoevolve/events.py +11 -0
  37. package/engine/autoevolve/git_workspace.py +77 -0
  38. package/engine/autoevolve/projection.py +23 -0
  39. package/engine/autoevolve/runner.py +413 -0
  40. package/engine/autoevolve/trust.py +146 -0
  41. package/engine/autoresearch/__init__.py +6 -0
  42. package/engine/autoresearch/commit.py +132 -0
  43. package/engine/autoresearch/contracts.py +126 -0
  44. package/engine/autoresearch/controller.py +207 -0
  45. package/engine/autoresearch/events.py +12 -0
  46. package/engine/autoresearch/gap_priority.py +168 -0
  47. package/engine/autoresearch/projection.py +30 -0
  48. package/engine/autoresearch/research_memory.py +59 -0
  49. package/engine/autoresearch/saturation.py +91 -0
  50. package/engine/briefs.py +2 -1
  51. package/engine/capabilities.py +1 -0
  52. package/engine/contracts.py +3 -1
  53. package/engine/evidencecore.py +7 -5
  54. package/engine/gaps.py +90 -51
  55. package/engine/judge_pack.py +65 -0
  56. package/engine/library_builtin.py +3 -1
  57. package/engine/living.py +2 -1
  58. package/engine/meta_synthesis.py +3 -1
  59. package/engine/orchestration.py +460 -0
  60. package/engine/pilot.py +2 -1
  61. package/engine/project.py +2 -2
  62. package/engine/research_service.py +113 -0
  63. package/engine/studio_read_model.py +400 -0
  64. package/engine/tribunal.py +1 -2
  65. package/engine/update.py +1 -0
  66. package/engine/versions.py +1 -1
  67. package/engine/worker_result.py +109 -0
  68. package/engine/workflows.py +70 -0
  69. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1720 -0
  70. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
  71. package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
  72. package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
  73. package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
  74. package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
  75. package/examples/ai-coding-assistant-evidence/frame.json +48 -0
  76. package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
  77. package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
  78. package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
  79. package/examples/ai-coding-assistant-evidence/report_spec.json +219 -0
  80. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2614 -0
  81. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2614 -0
  82. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2614 -0
  83. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2614 -0
  84. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2614 -0
  85. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1720 -0
  86. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1720 -0
  87. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1720 -0
  88. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1720 -0
  89. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1720 -0
  90. package/examples/ai-coding-assistant-evidence/result.json +1453 -0
  91. package/examples/ai-coding-assistant-evidence/result.zh.json +1453 -0
  92. package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
  93. package/examples/ai-coding-assistant-evidence/verdict.json +103 -0
  94. package/examples/workplace-ai-assistant/claims.jsonl +4 -0
  95. package/examples/workplace-ai-assistant/evaluation.json +19 -0
  96. package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
  97. package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
  98. package/examples/workplace-ai-assistant/frame.json +41 -0
  99. package/examples/workplace-ai-assistant/intervention.json +27 -0
  100. package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
  101. package/examples/workplace-ai-assistant/methodology.json +60 -0
  102. package/examples/workplace-ai-assistant/report_spec.json +55 -0
  103. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2484 -0
  104. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2484 -0
  105. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2484 -0
  106. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2484 -0
  107. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2484 -0
  108. package/examples/workplace-ai-assistant/result.json +553 -0
  109. package/examples/workplace-ai-assistant/result.zh.json +553 -0
  110. package/examples/workplace-ai-assistant/search_log.json +19 -0
  111. package/examples/workplace-ai-assistant/sources.jsonl +3 -0
  112. package/examples/workplace-ai-assistant/validation_result.json +9 -0
  113. package/examples/workplace-ai-assistant/verdict.json +52 -0
  114. package/install.sh +7 -7
  115. package/integrations/orchestration_dispatch.py +146 -0
  116. package/package.json +37 -3
  117. package/pyproject.toml +11 -20
  118. package/references/autoresearch.md +30 -0
  119. package/references/evaluation-policy.md +24 -0
  120. package/references/orchestration.md +22 -0
  121. package/references/scientific-invariants.md +19 -0
  122. package/retrieval/audit.py +154 -0
  123. package/schemas/intervention.schema.json +106 -21
  124. package/schemas/report-result.schema.json +9 -1
  125. package/schemas/v2/project.schema.json +2 -2
  126. package/schemas/v2/run.schema.json +1 -1
  127. package/schemas/vNext/autoevolve-session.schema.json +1 -0
  128. package/schemas/vNext/eval-snapshot.schema.json +1 -0
  129. package/schemas/vNext/execution-plan.schema.json +1 -0
  130. package/schemas/vNext/gap-priority.schema.json +1 -0
  131. package/schemas/vNext/negative-search-record.schema.json +1 -0
  132. package/schemas/vNext/research-iteration.schema.json +1 -0
  133. package/schemas/vNext/research-strategy.schema.json +1 -0
  134. package/schemas/vNext/skill-experiment.schema.json +1 -0
  135. package/schemas/vNext/task-spec.schema.json +1 -0
  136. package/schemas/vNext/worker-result.schema.json +1 -0
  137. package/scripts/benchmark_judge.py +2 -2
  138. package/scripts/benchmark_v3.py +26 -43
  139. package/scripts/build_esl_artifacts.py +2 -2
  140. package/scripts/build_evidence_library.py +2 -2
  141. package/scripts/build_gh_pages.py +98 -0
  142. package/scripts/build_readme_diagrams.py +72 -0
  143. package/scripts/build_report_variants.py +85 -0
  144. package/scripts/check_autoresearch_invariants.py +95 -0
  145. package/scripts/daily_evolve.py +30 -0
  146. package/scripts/dashboard_server.py +130 -101
  147. package/scripts/did_regression.py +5 -30
  148. package/scripts/enrich_projects_human_and_lieflat.py +1 -1
  149. package/scripts/generate_metrics.py +4 -3
  150. package/scripts/generate_new_projects.py +1 -1
  151. package/scripts/orchestrator.py +172 -18
  152. package/scripts/rebake_all_5themes.py +1 -2
  153. package/scripts/research_auto_cli.py +475 -0
  154. package/scripts/run_workspace.py +17 -7
  155. package/scripts/search_provenance.py +64 -0
  156. package/scripts/serve_web.py +9 -10
  157. package/scripts/skill_lint.py +1 -1
  158. package/scripts/skill_payload.py +78 -0
  159. package/scripts/validate_schema.py +15 -1
  160. package/scripts/vnext_cli.py +133 -0
  161. package/setup.py +12 -0
  162. package/skill/roles/registry.yaml +45 -0
  163. package/skill/sub-skills/report-generation/SKILL.md +12 -6
  164. package/skill/task-briefs/applicability.md +3 -0
  165. package/skill/task-briefs/projection.md +3 -0
  166. package/skill/workflows/decision-and-pilot.md +10 -0
  167. package/skill/workflows/evaluate-and-update.md +10 -0
  168. package/skill/workflows/evidence-review.md +13 -0
  169. package/visualization/eduevidence-report/assets/base.css +2 -2
  170. package/visualization/eduevidence-report/assets/reader.css +752 -0
  171. package/visualization/eduevidence-report/assets/reader.js +132 -0
  172. package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
  173. package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
  174. package/visualization/eduevidence-report/scripts/build_report.py +58 -65
  175. package/visualization/eduevidence-report/scripts/lieflat_engine.py +24 -100
  176. package/visualization/eduevidence-report/themes/academic.css +1 -1
  177. package/visualization/eduevidence-report/themes/claude.css +1 -1
  178. package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
  179. package/visualization/eduevidence-report/themes/datalab.css +2 -2
  180. package/visualization/eduevidence-report/themes/presentation.css +2 -2
  181. package/web/README.md +18 -0
  182. package/web/index.html +53 -0
  183. package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
  184. package/web/studio/assets/index-CzXocaGv.css +1 -0
  185. package/web/studio/assets/index-pa7jD7n4.js +230 -0
  186. package/web/studio/config.json +1 -0
  187. package/web/studio/index.html +14 -0
  188. package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
  189. package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
  190. package/engine/__pycache__/bias.cpython-312.pyc +0 -0
  191. package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
  192. package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
  193. package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
  194. package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
  195. package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
  196. package/engine/__pycache__/events.cpython-312.pyc +0 -0
  197. package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
  198. package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
  199. package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
  200. package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
  201. package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
  202. package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
  203. package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
  204. package/engine/__pycache__/ids.cpython-312.pyc +0 -0
  205. package/engine/__pycache__/library.cpython-312.pyc +0 -0
  206. package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
  207. package/engine/__pycache__/living.cpython-312.pyc +0 -0
  208. package/engine/__pycache__/log.cpython-312.pyc +0 -0
  209. package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
  210. package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
  211. package/engine/__pycache__/migration.cpython-312.pyc +0 -0
  212. package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
  213. package/engine/__pycache__/paths.cpython-312.pyc +0 -0
  214. package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
  215. package/engine/__pycache__/planner.cpython-312.pyc +0 -0
  216. package/engine/__pycache__/project.cpython-312.pyc +0 -0
  217. package/engine/__pycache__/projections.cpython-312.pyc +0 -0
  218. package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
  219. package/engine/__pycache__/run.cpython-312.pyc +0 -0
  220. package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
  221. package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
  222. package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
  223. package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
  224. package/engine/__pycache__/update.cpython-312.pyc +0 -0
  225. package/engine/__pycache__/versions.cpython-312.pyc +0 -0
  226. package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
  227. package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
  228. package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
  229. package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
  230. package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
  231. package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
  232. package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
  233. package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
  234. package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
  235. package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
  236. package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
  237. package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
  238. package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
  239. package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
  240. package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
  241. package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
  242. package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
  243. package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
  244. package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
  245. package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
  246. package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
  247. package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
  248. package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
  249. package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
  250. package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
  251. package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
  252. package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
  253. package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
  254. package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
  255. package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
  256. package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
  257. package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
  258. package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
  259. package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
  260. package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
  261. package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
  262. package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
  263. package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
  264. package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
  265. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
  266. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
  267. package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
  268. package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
  269. package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
  270. package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
  271. package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
  272. package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
  273. package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
  274. package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
  275. package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
  276. package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
@@ -0,0 +1,553 @@
1
+ {
2
+ "meta": {
3
+ "skill": "eduevidence",
4
+ "domain": "policy",
5
+ "version": "6.0.0",
6
+ "mode": "platform_native",
7
+ "question": "企业客服团队是否应引入生成式 AI 助手?",
8
+ "data_origin": "manual_curated"
9
+ },
10
+ "research_frame": {
11
+ "question": "企业客服团队是否应引入生成式 AI 助手?",
12
+ "decision_object": "adopt",
13
+ "intervention": {
14
+ "policy_name": "human_supervised_customer_support_assistant",
15
+ "policy_type": "institutional_reform",
16
+ "mechanism": "approved_knowledge_base_and_agent_review"
17
+ },
18
+ "population": {
19
+ "target_group": "enterprise_customer_support_staff",
20
+ "excluded_groups": "autonomous_agents_and_high_stakes_specialist_advice"
21
+ },
22
+ "comparison": "不提供生成式 AI 建议的现有客服流程。",
23
+ "outcomes": {
24
+ "primary": [
25
+ "policy_effectiveness",
26
+ "implementation_risk"
27
+ ],
28
+ "secondary": [
29
+ "cost_effectiveness",
30
+ "equity",
31
+ "feasibility"
32
+ ]
33
+ },
34
+ "context": {
35
+ "policy_environment": "enterprise_customer_support"
36
+ },
37
+ "scope": {
38
+ "time_range": "2023–2026",
39
+ "evidence_types": [
40
+ "quasi_experimental",
41
+ "rct"
42
+ ]
43
+ },
44
+ "success_condition": "提高每付薪工时的有效解决量,同时维持服务质量、隐私和员工自主判断。",
45
+ "extensions": {
46
+ "domain": "policy",
47
+ "data_origin": "manual_curated",
48
+ "note": "目的性人工选编,不是系统综述,也不属于模型执行 benchmark。"
49
+ }
50
+ },
51
+ "decision": {
52
+ "decision_question": "企业客服团队是否应引入生成式 AI 助手?",
53
+ "target_population": "企业客服人员,按资历和基线技能分层。",
54
+ "target_context": "使用经审核知识库、由人工监督的客服流程。",
55
+ "recommended_action": "pilot",
56
+ "confidence": "Low",
57
+ "confidence_score": null,
58
+ "independent_studies": 3,
59
+ "independent_samples": 3,
60
+ "supported_claims": [
61
+ "在有边界的客服场景中,AI 辅助可缩短会话处理时间;吞吐量是另一项指标。",
62
+ "ChatGPT 缩短了短篇职业写作任务用时;对客服而言属于间接证据。",
63
+ "在超出能力边界的任务中 AI 可能降低正确率;咨询实验对客服属于间接证据。",
64
+ "须单独监测资深高技能客服的质量,不能假定所有岗位群体均受益。"
65
+ ],
66
+ "uncertain_claims": [
67
+ "这组选编证据尚未确立本企业净成本、隐私事件率、亚组服务质量及持续表现。"
68
+ ],
69
+ "decision_rationale": "建议有监督试点:一项直接现场研究支持效率收益,但同时显示质量效应因人而异。间接实验提示任务边界,而本地安全性与净价值仍未知。",
70
+ "methodology_summary": "一项分批上线准实验和两项随机实验。只有客服研究属于直接证据;未计算合并标准化效应或模型 benchmark。",
71
+ "what_cannot_be_claimed": [
72
+ "普遍收益、自动部署安全、隐私保护、减员必要性或教学学习效果。"
73
+ ],
74
+ "missing_evidence": [
75
+ "这组选编证据尚未确立本企业净成本、隐私事件率、亚组服务质量及持续表现。"
76
+ ],
77
+ "applicability": {
78
+ "required_conditions": [
79
+ "从低风险且知识库覆盖的队列开始;客服审查并编辑每条建议回复。",
80
+ "按资历和基线技能分层;保留专家判断并记录复核工作量。",
81
+ "试点前落实客户数据最小化与脱敏、访问和留存限制,并审查供应商数据使用条款。",
82
+ "陌生、模糊、敏感或高风险请求交由合格人员处理;禁止自动作出承诺。"
83
+ ]
84
+ },
85
+ "extensions": {
86
+ "data_origin": "manual_curated",
87
+ "benchmark_eligible": false,
88
+ "note": "针对全面部署的保守人工判断;不是确定性置信度策略运行结果,也不是概率。",
89
+ "knowledge_gaps": [
90
+ {
91
+ "gap_id": "G-001",
92
+ "evidence_ids": [
93
+ "E-001",
94
+ "E-002",
95
+ "E-003",
96
+ "E-004"
97
+ ],
98
+ "summary": "这组选编证据尚未确立本企业净成本、隐私事件率、亚组服务质量及持续表现。"
99
+ }
100
+ ]
101
+ }
102
+ },
103
+ "sources": [
104
+ {
105
+ "source_id": "S-001",
106
+ "title": "Generative AI at Work",
107
+ "authors": [
108
+ "Erik Brynjolfsson",
109
+ "Danielle Li",
110
+ "Lindsey Raymond"
111
+ ],
112
+ "year": 2025,
113
+ "doi": "10.1093/qje/qjae044",
114
+ "canonical_url": "https://academic.oup.com/qje/article/140/2/889/7990658",
115
+ "source_type": "journal_article",
116
+ "authority_level": "tier1_paper_doi",
117
+ "status": "VALID",
118
+ "source_locator": {
119
+ "type": "section",
120
+ "section": "Abstract; III.B and Table I; IV.A and Table II; IV.B; VIII"
121
+ },
122
+ "fetch": {
123
+ "original_url": "https://academic.oup.com/qje/article/140/2/889/7990658",
124
+ "fetch_method": "native",
125
+ "fetch_provider": "builtin",
126
+ "fetch_status": "FETCH_VALID"
127
+ },
128
+ "extensions": {
129
+ "verified_on": "2026-09-08",
130
+ "data_origin": "manual_curated",
131
+ "version": "published QJE article",
132
+ "verification_scope": "Full primary text read through web tool; no Crossref registry or retraction check claimed."
133
+ }
134
+ },
135
+ {
136
+ "source_id": "S-002",
137
+ "title": "Experimental evidence on the productivity effects of generative artificial intelligence",
138
+ "authors": [
139
+ "Shakked Noy",
140
+ "Whitney Zhang"
141
+ ],
142
+ "year": 2023,
143
+ "doi": "10.1126/science.adh2586",
144
+ "canonical_url": "https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf",
145
+ "source_type": "journal_article",
146
+ "authority_level": "tier1_paper_doi",
147
+ "status": "VALID",
148
+ "source_locator": {
149
+ "type": "section",
150
+ "section": "Author manuscript pp. 1, 3–4, 7; Figure 1 (p. 12)"
151
+ },
152
+ "fetch": {
153
+ "original_url": "https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf",
154
+ "fetch_method": "native",
155
+ "fetch_provider": "builtin",
156
+ "fetch_status": "FETCH_VALID"
157
+ },
158
+ "extensions": {
159
+ "verified_on": "2026-09-08",
160
+ "data_origin": "manual_curated",
161
+ "version": "author manuscript associated with Science 2023; not the March 444-person draft",
162
+ "verification_scope": "Full primary text read through web tool; no Crossref registry or retraction check claimed."
163
+ }
164
+ },
165
+ {
166
+ "source_id": "S-003",
167
+ "title": "Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality",
168
+ "authors": [
169
+ "Fabrizio Dell’Acqua",
170
+ "Edward McFowland III",
171
+ "Ethan Mollick",
172
+ "Hila Lifshitz-Assaf",
173
+ "Katherine C. Kellogg",
174
+ "Saran Rajendran",
175
+ "Lisa Krayer",
176
+ "François Candelon",
177
+ "Karim R. Lakhani"
178
+ ],
179
+ "year": 2026,
180
+ "doi": "10.1287/orsc.2025.21838",
181
+ "canonical_url": "https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838",
182
+ "source_type": "journal_article",
183
+ "authority_level": "tier1_paper_doi",
184
+ "status": "VALID",
185
+ "source_locator": {
186
+ "type": "section",
187
+ "section": "Sections 3 and 4.2; Figure 5; Table 7"
188
+ },
189
+ "fetch": {
190
+ "original_url": "https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838",
191
+ "fetch_method": "native",
192
+ "fetch_provider": "builtin",
193
+ "fetch_status": "FETCH_VALID"
194
+ },
195
+ "extensions": {
196
+ "verified_on": "2026-09-08",
197
+ "data_origin": "manual_curated",
198
+ "version": "published Organization Science 2026 article",
199
+ "verification_scope": "Full primary text read through web tool; no Crossref registry or retraction check claimed."
200
+ }
201
+ }
202
+ ],
203
+ "evidence": [
204
+ {
205
+ "evidence_id": "E-001",
206
+ "source_id": "S-001",
207
+ "study_id": "ST-001",
208
+ "sample_id": "SAMPLE-support-all",
209
+ "claim_id": "C-001",
210
+ "title": "Generative AI at Work",
211
+ "year": 2025,
212
+ "study_type": "quasi_experimental",
213
+ "population": "客服人员",
214
+ "sample_size": 5172,
215
+ "outcome_type": "completion_time",
216
+ "outcome_measure": "会话处理时间;另列吞吐量摘要:每小时解决问题数约增加 15%。",
217
+ "claim": "在有边界的客服场景中,AI 辅助可缩短会话处理时间;吞吐量是另一项指标。",
218
+ "direction": "support",
219
+ "relation_to_claim": "support",
220
+ "effect_direction": "positive",
221
+ "decision_relation": "conditional",
222
+ "source_location": "https://academic.oup.com/qje/article/140/2/889/7990658",
223
+ "limitations": [
224
+ "单一企业、单一工具、非随机分批上线;因果解释依赖识别假设。"
225
+ ],
226
+ "status": "SUPPORTED",
227
+ "extensions": {
228
+ "domain": "policy",
229
+ "policy_outcome": "policy_effectiveness",
230
+ "directness": "direct",
231
+ "raw_result": {
232
+ "metric": "issues_resolved_per_hour_relative_change",
233
+ "value": 15,
234
+ "unit": "percent",
235
+ "role": "separate_throughput_measure_not_completion_time",
236
+ "ci_lower": null,
237
+ "ci_upper": null,
238
+ "p_value": null,
239
+ "uncertainty_status": "not_extracted_for_this_summary_estimand"
240
+ },
241
+ "standardized_effect": null,
242
+ "study_total_n": 5172,
243
+ "note": "按条目注明样本;研究总人数不得替代分析样本。E-004 属于 S-001 同一队列,不另计独立样本。"
244
+ }
245
+ },
246
+ {
247
+ "evidence_id": "E-002",
248
+ "source_id": "S-002",
249
+ "study_id": "ST-002",
250
+ "sample_id": "SAMPLE-writing",
251
+ "claim_id": "C-002",
252
+ "title": "Experimental evidence on the productivity effects of generative artificial intelligence",
253
+ "year": 2023,
254
+ "study_type": "rct",
255
+ "population": "受过大学教育的职场人士",
256
+ "sample_size": 453,
257
+ "outcome_type": "completion_time",
258
+ "outcome_measure": "自报任务用时约减少 40%;评定写作质量约提高 18% 是另一项指标。",
259
+ "claim": "ChatGPT 缩短了短篇职业写作任务用时;对客服而言属于间接证据。",
260
+ "direction": "support",
261
+ "relation_to_claim": "support",
262
+ "effect_direction": "positive",
263
+ "decision_relation": "conditional",
264
+ "source_location": "https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf",
265
+ "limitations": [
266
+ "间接证据,不可直接推广:短时激励写作任务不要求精确事实或客户特定情境。"
267
+ ],
268
+ "status": "SUPPORTED",
269
+ "extensions": {
270
+ "domain": "policy",
271
+ "policy_outcome": "policy_effectiveness",
272
+ "directness": "indirect",
273
+ "raw_result": {
274
+ "metric": "task_time_relative_change",
275
+ "value": -40,
276
+ "unit": "percent",
277
+ "ci_lower": null,
278
+ "ci_upper": null,
279
+ "p_value": null,
280
+ "uncertainty_status": "not_extracted_for_this_summary_estimand"
281
+ },
282
+ "standardized_effect": null,
283
+ "study_total_n": 453,
284
+ "note": "按条目注明样本;研究总人数不得替代分析样本。E-004 属于 S-001 同一队列,不另计独立样本。"
285
+ }
286
+ },
287
+ {
288
+ "evidence_id": "E-003",
289
+ "source_id": "S-003",
290
+ "study_id": "ST-003",
291
+ "sample_id": "SAMPLE-consulting-outside",
292
+ "claim_id": "C-003",
293
+ "title": "Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality",
294
+ "year": 2026,
295
+ "study_type": "rct",
296
+ "population": "参与超出能力边界实验的 BCG 顾问",
297
+ "sample_size": 373,
298
+ "outcome_type": "accuracy",
299
+ "outcome_measure": "商业建议正确率:合并 AI 组约低 19 个百分点;表 7 为 373 人,不是 758 人。",
300
+ "claim": "在超出能力边界的任务中 AI 可能降低正确率;咨询实验对客服属于间接证据。",
301
+ "direction": "support",
302
+ "relation_to_claim": "support",
303
+ "effect_direction": "negative",
304
+ "decision_relation": "conditional",
305
+ "source_location": "https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838",
306
+ "limitations": [
307
+ "间接证据,不可直接推广:咨询顾问处理实验商业案例,并非真实客服工单。"
308
+ ],
309
+ "status": "SUPPORTED",
310
+ "extensions": {
311
+ "domain": "policy",
312
+ "policy_outcome": "implementation_risk",
313
+ "directness": "indirect",
314
+ "raw_result": {
315
+ "metric": "correctness_absolute_change",
316
+ "value": -19,
317
+ "unit": "percentage_points",
318
+ "ci_lower": null,
319
+ "ci_upper": null,
320
+ "p_value": null,
321
+ "uncertainty_status": "not_extracted_for_this_summary_estimand"
322
+ },
323
+ "standardized_effect": null,
324
+ "study_total_n": 758,
325
+ "note": "按条目注明样本;研究总人数不得替代分析样本。E-004 属于 S-001 同一队列,不另计独立样本。"
326
+ }
327
+ },
328
+ {
329
+ "evidence_id": "E-004",
330
+ "source_id": "S-001",
331
+ "study_id": "ST-001",
332
+ "sample_id": "SAMPLE-support-all",
333
+ "claim_id": "C-004",
334
+ "title": "Generative AI at Work",
335
+ "year": 2025,
336
+ "study_type": "quasi_experimental",
337
+ "population": "资深高技能客服人员",
338
+ "sample_size": null,
339
+ "outcome_type": "accuracy",
340
+ "outcome_measure": "资历最深、技能最高的客服群体出现小幅质量下降。",
341
+ "claim": "须单独监测资深高技能客服的质量,不能假定所有岗位群体均受益。",
342
+ "direction": "support",
343
+ "relation_to_claim": "support",
344
+ "effect_direction": "negative",
345
+ "decision_relation": "conditional",
346
+ "source_location": "https://academic.oup.com/qje/article/140/2/889/7990658",
347
+ "limitations": [
348
+ "与 E-001 属于同一研究;未提取该亚组样本量,不是独立重复验证。"
349
+ ],
350
+ "status": "SUPPORTED",
351
+ "extensions": {
352
+ "domain": "policy",
353
+ "policy_outcome": "implementation_risk",
354
+ "directness": "direct",
355
+ "raw_result": {
356
+ "metric": "experienced_staff_quality",
357
+ "value": null,
358
+ "unit": "not_extracted",
359
+ "ci_lower": null,
360
+ "ci_upper": null,
361
+ "p_value": null,
362
+ "uncertainty_status": "not_extracted_for_this_summary_estimand"
363
+ },
364
+ "standardized_effect": null,
365
+ "study_total_n": 5172,
366
+ "note": "按条目注明样本;研究总人数不得替代分析样本。E-004 属于 S-001 同一队列,不另计独立样本。"
367
+ }
368
+ }
369
+ ],
370
+ "claims": [
371
+ {
372
+ "claim_id": "C-001",
373
+ "claim": "在有边界的客服场景中,AI 辅助可缩短会话处理时间;吞吐量是另一项指标。",
374
+ "outcome_type": "completion_time",
375
+ "evidence_ids": [
376
+ "E-001"
377
+ ],
378
+ "status": "SUPPORTED",
379
+ "pooled_effect_g": null
380
+ },
381
+ {
382
+ "claim_id": "C-002",
383
+ "claim": "ChatGPT 缩短了短篇职业写作任务用时;对客服而言属于间接证据。",
384
+ "outcome_type": "completion_time",
385
+ "evidence_ids": [
386
+ "E-002"
387
+ ],
388
+ "status": "SUPPORTED",
389
+ "pooled_effect_g": null
390
+ },
391
+ {
392
+ "claim_id": "C-003",
393
+ "claim": "在超出能力边界的任务中 AI 可能降低正确率;咨询实验对客服属于间接证据。",
394
+ "outcome_type": "accuracy",
395
+ "evidence_ids": [
396
+ "E-003"
397
+ ],
398
+ "status": "SUPPORTED",
399
+ "pooled_effect_g": null
400
+ },
401
+ {
402
+ "claim_id": "C-004",
403
+ "claim": "须单独监测资深高技能客服的质量,不能假定所有岗位群体均受益。",
404
+ "outcome_type": "accuracy",
405
+ "evidence_ids": [
406
+ "E-004"
407
+ ],
408
+ "status": "SUPPORTED",
409
+ "pooled_effect_g": null
410
+ }
411
+ ],
412
+ "outcomes": [
413
+ {
414
+ "outcome_type": "completion_time",
415
+ "positive_count": 2,
416
+ "negative_count": 0,
417
+ "null_count": 0,
418
+ "evidence_ids": [
419
+ "E-001",
420
+ "E-002"
421
+ ]
422
+ },
423
+ {
424
+ "outcome_type": "accuracy",
425
+ "positive_count": 0,
426
+ "negative_count": 2,
427
+ "null_count": 0,
428
+ "evidence_ids": [
429
+ "E-003",
430
+ "E-004"
431
+ ]
432
+ }
433
+ ],
434
+ "methodology_reviews": [
435
+ {
436
+ "target": "overall",
437
+ "verdict": "CONCERN",
438
+ "audit_items": {
439
+ "evidence_level": {
440
+ "status": "met"
441
+ },
442
+ "causal_identification": {
443
+ "status": "partial"
444
+ },
445
+ "external_validity": {
446
+ "status": "partial"
447
+ },
448
+ "cost_evidence": {
449
+ "status": "missing"
450
+ },
451
+ "stakeholder_representation": {
452
+ "status": "partial"
453
+ },
454
+ "implementation_evidence": {
455
+ "status": "partial"
456
+ },
457
+ "equity_analysis": {
458
+ "status": "partial"
459
+ },
460
+ "effect_size_reported": {
461
+ "status": "partial"
462
+ },
463
+ "uncertainty_quantified": {
464
+ "status": "partial"
465
+ },
466
+ "comparator_clarity": {
467
+ "status": "met"
468
+ },
469
+ "publication_bias_risk": {
470
+ "status": "partial"
471
+ },
472
+ "generalizability_claims": {
473
+ "status": "met"
474
+ }
475
+ },
476
+ "task_vs_learning_guard": {
477
+ "measured_construct": "workplace_task_performance",
478
+ "equates_task_with_learning": false,
479
+ "note": "教学结果不适用;生产率并非学生学习效果主张。"
480
+ },
481
+ "limitations": [
482
+ "单一企业、单一工具、非随机分批上线;因果解释依赖识别假设。",
483
+ "间接证据,不可直接推广:短时激励写作任务不要求精确事实或客户特定情境。",
484
+ "间接证据,不可直接推广:咨询顾问处理实验商业案例,并非真实客服工单。",
485
+ "与 E-001 属于同一研究;未提取该亚组样本量,不是独立重复验证。"
486
+ ],
487
+ "suggestions": [
488
+ "从低风险且知识库覆盖的队列开始;客服审查并编辑每条建议回复。",
489
+ "按资历和基线技能分层;保留专家判断并记录复核工作量。",
490
+ "试点前落实客户数据最小化与脱敏、访问和留存限制,并审查供应商数据使用条款。",
491
+ "陌生、模糊、敏感或高风险请求交由合格人员处理;禁止自动作出承诺。"
492
+ ],
493
+ "extensions": {}
494
+ }
495
+ ],
496
+ "applicability": {
497
+ "suitable_for": "从低风险且知识库覆盖的队列开始;客服审查并编辑每条建议回复。",
498
+ "required_conditions": [
499
+ "从低风险且知识库覆盖的队列开始;客服审查并编辑每条建议回复。",
500
+ "按资历和基线技能分层;保留专家判断并记录复核工作量。",
501
+ "试点前落实客户数据最小化与脱敏、访问和留存限制,并审查供应商数据使用条款。",
502
+ "陌生、模糊、敏感或高风险请求交由合格人员处理;禁止自动作出承诺。"
503
+ ],
504
+ "not_suitable_for": "普遍收益、自动部署安全、隐私保护、减员必要性或教学学习效果。"
505
+ },
506
+ "intervention": {
507
+ "decision": "pilot",
508
+ "target_population": "企业客服人员,按资历和基线技能分层。",
509
+ "pilot_duration": "建议:两周基线和六周试点;尚未执行。",
510
+ "ai_usage_policy": "从低风险且知识库覆盖的队列开始;客服审查并编辑每条建议回复。",
511
+ "risk_control": [
512
+ "从低风险且知识库覆盖的队列开始;客服审查并编辑每条建议回复。",
513
+ "按资历和基线技能分层;保留专家判断并记录复核工作量。",
514
+ "试点前落实客户数据最小化与脱敏、访问和留存限制,并审查供应商数据使用条款。",
515
+ "陌生、模糊、敏感或高风险请求交由合格人员处理;禁止自动作出承诺。"
516
+ ],
517
+ "stop_conditions": [
518
+ "一旦核实隐私泄漏、严重不安全建议或越权客户承诺,立即暂停,调查后再决定是否恢复。",
519
+ "若盲评质量越过预先约定的非劣界值(包括各资历分层),暂停扩展。"
520
+ ],
521
+ "evidence_alignment": [
522
+ "E-001",
523
+ "E-002",
524
+ "E-003",
525
+ "E-004"
526
+ ],
527
+ "extensions": {
528
+ "gap_id": "G-001",
529
+ "target_population": "enterprise_customer_support_staff",
530
+ "status": "proposed_not_executed"
531
+ }
532
+ },
533
+ "evaluation": {
534
+ "research_question": "企业客服团队是否应引入生成式 AI 助手?",
535
+ "groups": {
536
+ "treatment": "符合条件的团队按资历与基线表现分层,随机分配有监督助手使用权限。",
537
+ "comparison": "同期维持现有流程的团队;记录组间污染。"
538
+ },
539
+ "baseline": "分配前记录解决率、付薪工时、重复联系、盲评质量及复核成本。",
540
+ "post_test": "试点结束时评估相同结果;独立审计隐私与不安全承诺。",
541
+ "analysis_plan": "入组前预注册意向治疗比较、团队聚类不确定性及亚组估计。根据本地基线变异和业务目标确定样本量与非劣界值。",
542
+ "success_threshold": "仅在质量非劣、每付薪工时有效解决量提高、净成本可接受且无未解决严重安全事件时考虑扩展;阈值须在本地预先约定。",
543
+ "stop_conditions": [
544
+ "一旦核实隐私泄漏、严重不安全建议或越权客户承诺,立即暂停,调查后再决定是否恢复。",
545
+ "若盲评质量越过预先约定的非劣界值(包括各资历分层),暂停扩展。"
546
+ ],
547
+ "extensions": {
548
+ "gap_id": "G-001",
549
+ "status": "proposed_not_executed"
550
+ }
551
+ },
552
+ "forest_plot_data": []
553
+ }
@@ -0,0 +1,19 @@
1
+ {
2
+ "data_origin": "manual_curated",
3
+ "checked_on": "2026-09-08",
4
+ "scope": "Purposive primary-source review for a bounded organizational-policy demo; not a systematic search or model benchmark",
5
+ "plan": ["Prioritize field evidence from customer support", "Check task-boundary counterevidence from randomized workplace experiments", "Read original text and distinguish publication versions and analyzed samples"],
6
+ "queries": ["site.academic.oup.com Generative AI at Work 5172 15%", "site.science.org Noy Zhang experimental evidence productivity generative artificial intelligence 453", "site.hbs.edu Dell Acqua navigating jagged technological frontier 758 19 percentage points pdf", "site.hbs.edu 24-013 pdf", "site.shakkednoy.com science pdf productivity"],
7
+ "included": ["S-001", "S-002", "S-003"],
8
+ "attempts": [
9
+ {"url": "https://academic.oup.com/qje/article/140/2/889/7990658", "result": "Full publisher text read; redirected to https://oup.silverchair-cdn.com/article-minimal/7990658"},
10
+ {"url": "https://www.science.org/doi/10.1126/science.adh2586", "result": "Web tool internal error; no successful publisher fetch claimed"},
11
+ {"url": "https://economics.mit.edu/sites/default/files/inline-files/Noy_Zhang_1.pdf", "result": "Read early March 2023 working paper with 444 participants; excluded to avoid mixing versions"},
12
+ {"url": "https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf", "result": "Read author manuscript, including Method and Limitations; 453 participants"},
13
+ {"url": "https://www.hbs.edu/ris/Publication%20Files/24-013_d9b45b68-9e74-42d6-a1c6-c72fb70c7282.pdf", "result": "Web tool internal error; used published primary text instead"},
14
+ {"url": "https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838", "result": "Read published 2026 text, section 4.2, Figure 5 and Table 7; outside-frontier sample 373"}
15
+ ],
16
+ "counterevidence": "S-003 outside-frontier correctness decline; S-001 experienced-worker quality heterogeneity",
17
+ "excluded_discovery_results": "Generated research assessments and secondary summaries were not used as evidence",
18
+ "stopping_reason": "Three verified primary studies meet demo scope; only one is directly in customer support. No claim of exhaustive retrieval or absence of other null findings."
19
+ }
@@ -0,0 +1,3 @@
1
+ {"source_id": "S-001", "title": "Generative AI at Work", "authors": ["Erik Brynjolfsson", "Danielle Li", "Lindsey Raymond"], "year": 2025, "doi": "10.1093/qje/qjae044", "canonical_url": "https://academic.oup.com/qje/article/140/2/889/7990658", "source_type": "journal_article", "authority_level": "tier1_paper_doi", "status": "VALID", "source_locator": {"type": "section", "section": "Abstract; III.B and Table I; IV.A and Table II; IV.B; VIII"}, "fetch": {"original_url": "https://academic.oup.com/qje/article/140/2/889/7990658", "fetch_method": "native", "fetch_provider": "builtin", "fetch_status": "FETCH_VALID"}, "extensions": {"verified_on": "2026-09-08", "data_origin": "manual_curated", "version": "published QJE article", "verification_scope": "Full primary text read through web tool; no Crossref registry or retraction check claimed."}}
2
+ {"source_id": "S-002", "title": "Experimental evidence on the productivity effects of generative artificial intelligence", "authors": ["Shakked Noy", "Whitney Zhang"], "year": 2023, "doi": "10.1126/science.adh2586", "canonical_url": "https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf", "source_type": "journal_article", "authority_level": "tier1_paper_doi", "status": "VALID", "source_locator": {"type": "section", "section": "Author manuscript pp. 1, 3–4, 7; Figure 1 (p. 12)"}, "fetch": {"original_url": "https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf", "fetch_method": "native", "fetch_provider": "builtin", "fetch_status": "FETCH_VALID"}, "extensions": {"verified_on": "2026-09-08", "data_origin": "manual_curated", "version": "author manuscript associated with Science 2023; not the March 444-person draft", "verification_scope": "Full primary text read through web tool; no Crossref registry or retraction check claimed."}}
3
+ {"source_id": "S-003", "title": "Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality", "authors": ["Fabrizio Dell’Acqua", "Edward McFowland III", "Ethan Mollick", "Hila Lifshitz-Assaf", "Katherine C. Kellogg", "Saran Rajendran", "Lisa Krayer", "François Candelon", "Karim R. Lakhani"], "year": 2026, "doi": "10.1287/orsc.2025.21838", "canonical_url": "https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838", "source_type": "journal_article", "authority_level": "tier1_paper_doi", "status": "VALID", "source_locator": {"type": "section", "section": "Sections 3 and 4.2; Figure 5; Table 7"}, "fetch": {"original_url": "https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838", "fetch_method": "native", "fetch_provider": "builtin", "fetch_status": "FETCH_VALID"}, "extensions": {"verified_on": "2026-09-08", "data_origin": "manual_curated", "version": "published Organization Science 2026 article", "verification_scope": "Full primary text read through web tool; no Crossref registry or retraction check claimed."}}
@@ -0,0 +1,9 @@
1
+ {
2
+ "status": "PASS",
3
+ "schema_objects_validated": 27,
4
+ "graph_roundtrip": "PASS",
5
+ "report_input_gates": "PASS",
6
+ "studio_scanned": "workplace-ai-assistant",
7
+ "studio_domain": "education",
8
+ "reports_generated": false
9
+ }
@@ -0,0 +1,52 @@
1
+ {
2
+ "decision_question": "Should an enterprise customer-support team introduce a generative AI assistant?",
3
+ "target_population": "Enterprise customer-support staff, stratified by tenure and baseline skill.",
4
+ "target_context": "Human-supervised support using an approved knowledge base.",
5
+ "recommended_action": "pilot",
6
+ "confidence": "Low",
7
+ "confidence_score": null,
8
+ "independent_studies": 3,
9
+ "independent_samples": 3,
10
+ "supported_claims": [
11
+ "AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.",
12
+ "ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.",
13
+ "AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.",
14
+ "Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits."
15
+ ],
16
+ "uncertain_claims": [
17
+ "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set."
18
+ ],
19
+ "decision_rationale": "A supervised pilot is warranted because one direct field study supports efficiency gains but also reveals heterogeneous quality effects. Indirect experiments identify task boundaries, while local safety and net value remain unknown.",
20
+ "methodology_summary": "One staggered-rollout quasi-experiment and two randomized experiments. Only the support study is direct; no pooled standardized effect or model benchmark was computed.",
21
+ "what_cannot_be_claimed": [
22
+ "Universal gains, autonomous deployment safety, privacy protection, reduced staffing requirements or educational learning gains."
23
+ ],
24
+ "missing_evidence": [
25
+ "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set."
26
+ ],
27
+ "applicability": {
28
+ "required_conditions": [
29
+ "Start with low-risk, knowledge-base-covered queues; agents inspect and edit every proposed reply.",
30
+ "Stratify by tenure and baseline skill; retain expert discretion and measure review workload.",
31
+ "Minimize and redact customer data, restrict access and retention, and review supplier data-use terms before a pilot.",
32
+ "Escalate unfamiliar, ambiguous, sensitive or high-stakes requests to qualified humans; no autonomous commitments."
33
+ ]
34
+ },
35
+ "extensions": {
36
+ "data_origin": "manual_curated",
37
+ "benchmark_eligible": false,
38
+ "note": "Conservative manual judgment about broad deployment; not a deterministic confidence-policy run or a probability.",
39
+ "knowledge_gaps": [
40
+ {
41
+ "gap_id": "G-001",
42
+ "evidence_ids": [
43
+ "E-001",
44
+ "E-002",
45
+ "E-003",
46
+ "E-004"
47
+ ],
48
+ "summary": "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set."
49
+ }
50
+ ]
51
+ }
52
+ }