eduevidence 6.2.0 → 6.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +395 -0
- package/README.md +22 -13
- package/README.zh-CN.md +15 -8
- package/SKILL.md +10 -9
- package/benchmarks/evidence-library.json +277 -1
- package/docs/architecture.md +6 -3
- package/docs/j-ev-experimental.md +250 -0
- package/docs/reproducibility.md +138 -0
- package/domains/_neutral/copy/few_shots.json +21 -0
- package/domains/_neutral/copy/framing_lexicon.json +19 -0
- package/domains/_neutral/copy/module_labels.json +5 -0
- package/domains/_neutral/copy/module_labels_footer.json +102 -0
- package/domains/_neutral/copy/module_labels_modules.json +204 -0
- package/domains/_neutral/copy/module_labels_nav.json +126 -0
- package/domains/_neutral/copy/module_labels_summary.json +98 -0
- package/domains/_neutral/copy/module_labels_tables.json +164 -0
- package/domains/_neutral/copy/module_labels_v2.json +90 -0
- package/domains/_neutral/copy/risk_constructs.json +20 -0
- package/domains/_neutral/copy/section_titles.json +66 -0
- package/domains/_neutral/copy/terminology.json +11 -0
- package/domains/check_copy_packs.py +103 -0
- package/domains/education/copy/few_shots.json +22 -0
- package/domains/education/copy/framing_enums.json +167 -0
- package/domains/education/copy/framing_lexicon.json +166 -0
- package/domains/education/copy/module_labels.json +169 -0
- package/domains/education/copy/risk_constructs.json +48 -0
- package/domains/education/copy/section_titles.json +186 -0
- package/domains/education/copy/terminology.json +70 -0
- package/domains/education/manifest.json +1 -1
- package/domains/education/outcome_taxonomy.json +2 -2
- package/domains/manifest.json +1 -1
- package/domains/policy/copy/few_shots.json +22 -0
- package/domains/policy/copy/framing_enums.json +94 -0
- package/domains/policy/copy/framing_lexicon.json +174 -0
- package/domains/policy/copy/module_labels.json +168 -0
- package/domains/policy/copy/risk_constructs.json +33 -0
- package/domains/policy/copy/section_titles.json +186 -0
- package/domains/policy/copy/terminology.json +64 -0
- package/engine/capabilities.py +57 -5
- package/engine/decision_policy.py +88 -17
- package/engine/library_builtin.py +7 -4
- package/engine/tribunal.py +17 -23
- package/engine/versions.py +1 -1
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +4 -4
- package/examples/spaced-retrieval-practice/EduEvidence_Report.html +2728 -0
- package/examples/spaced-retrieval-practice/report.html +2522 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +4 -4
- package/examples/workplace-ai-assistant/EduEvidence_Report.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +36 -36
- package/integrations/jev/__init__.py +115 -0
- package/integrations/jev/approval.py +212 -0
- package/integrations/jev/cli.py +84 -0
- package/integrations/jev/config.py +112 -0
- package/integrations/jev/gateway.py +128 -0
- package/integrations/jev/modes.py +38 -0
- package/integrations/jev/tools_classify.py +88 -0
- package/integrations/jev/tools_extract.py +111 -0
- package/integrations/jev/tools_rerank.py +71 -0
- package/integrations/jev/tools_screen.py +87 -0
- package/integrations/jev/tools_verify.py +95 -0
- package/integrations/jev_mcp.py +22 -0
- package/integrations/semantic_decide.py +286 -0
- package/integrations/semdecide_cli.py +55 -0
- package/package.json +9 -1
- package/pyproject.toml +1 -1
- package/references/report-copy-style.md +43 -3
- package/schemas/v2/decision-snapshot.schema.json +20 -9
- package/schemas/v2/intake.schema.json +191 -0
- package/scripts/build_evidence_library.py +15 -5
- package/scripts/dashboard_server.py +13 -2
- package/scripts/intake/__init__.py +31 -0
- package/scripts/intake/__main__.py +18 -0
- package/scripts/intake/background.py +78 -0
- package/scripts/intake/browser.py +79 -0
- package/scripts/intake/cli.py +57 -0
- package/scripts/intake/constants.py +57 -0
- package/scripts/intake/depth.py +53 -0
- package/scripts/intake/enhancements.py +106 -0
- package/scripts/intake/hooks.py +90 -0
- package/scripts/intake/prefs.py +76 -0
- package/scripts/intake/prompts.py +85 -0
- package/scripts/intake/session.py +152 -0
- package/scripts/lint_file_layers.py +126 -0
- package/scripts/orchestrator.py +68 -17
- package/scripts/pre_verdict_gate.py +21 -7
- package/scripts/skill_lint.py +11 -1
- package/scripts/skill_payload.py +3 -3
- package/scripts/test_adversarial_empirical.py +70 -6
- package/skill/agents/evidence-judge.md +49 -7
- package/skill/workflows/experimental-jev.md +170 -0
- package/skill/workflows/intake.md +120 -0
- package/visualization/eduevidence-report/scripts/build_infographics.py +32 -14
- package/visualization/eduevidence-report/scripts/build_report.py +75 -662
- package/visualization/eduevidence-report/scripts/report_copy_pack.py +296 -0
- package/visualization/eduevidence-report/scripts/report_copy_policy_guard.py +47 -0
- package/visualization/eduevidence-report/scripts/zh_labels.py +61 -0
- package/scripts/build_esl_artifacts.py +0 -1921
- package/scripts/build_killer_demo.py +0 -295
- package/scripts/enrich_projects_human_and_lieflat.py +0 -315
- package/scripts/generate_new_projects.py +0 -686
- package/scripts/sync_killer_demo_report.py +0 -270
|
@@ -1911,7 +1911,7 @@ select:focus-visible,
|
|
|
1911
1911
|
<button type="button" class="report-view-btn" data-report-view="full" aria-pressed="false">完整报告</button>
|
|
1912
1912
|
</nav>
|
|
1913
1913
|
</header>
|
|
1914
|
-
<div class="report-page report-page-brief" data-report-page="brief"><nav class="brief-navigation" aria-label="摘要导航"><a href="#brief-decision-zh"><span>01</span>先看结论</a><a href="#brief-outcomes-zh"><span>02</span
|
|
1914
|
+
<div class="report-page report-page-brief" data-report-page="brief"><nav class="brief-navigation" aria-label="摘要导航"><a href="#brief-decision-zh"><span>01</span>先看结论</a><a href="#brief-outcomes-zh"><span>02</span>过程产出 ≠ 政策效果</a><a href="#brief-tribunal-zh"><span>03</span>证据裁决</a><a href="#brief-action-zh"><span>04</span>从证据到行动</a><a href="#brief-sources-zh"><span>05</span>关键来源</a></nav><div class="brief-reading-content"><section class="brief-block brief-decision" id="brief-decision-zh"><header class="brief-block-header"><h2>先看结论</h2><p>该不该做、置信度多高、最关键的证据边界在哪。</p></header><div class="brief-block-body">
|
|
1915
1915
|
<div class="decision-hero pilot" data-visual="decision-hero">
|
|
1916
1916
|
<div class="hero-decision">
|
|
1917
1917
|
<span class="eyebrow">建议决策</span>
|
|
@@ -1926,7 +1926,7 @@ select:focus-visible,
|
|
|
1926
1926
|
<article class="hero-insight next"><span>下一步</span><p class="hero-insight-text">在已批准知识库上开展人工监督试点,每条回复均经坐席复核,并跟踪升级率与纠错率。</p></article>
|
|
1927
1927
|
</div>
|
|
1928
1928
|
<p class="hero-provenance"><span>证据 / 来源</span> · 4 / 3</p>
|
|
1929
|
-
</div></div></section><section class="brief-block brief-outcomes" id="brief-outcomes-zh"><header class="brief-block-header"><h2
|
|
1929
|
+
</div></div></section><section class="brief-block brief-outcomes" id="brief-outcomes-zh"><header class="brief-block-header"><h2>过程产出 ≠ 政策效果</h2><p>只展示真正有解释力的结果分离;正向、负向与零效应按 effect_direction 编码。</p></header><div class="brief-block-body"><div class="outcome-separation" data-visual="outcome-separation"><div class="visual-heading"><h3>结果分离 · 过程产出 ≠ 政策效果</h3><p>将不同结果类型分开裁决,避免把试点期产出或效率提升直接等同于真实政策效果。</p><p class="semantic-note">效应方向来自 evidence.effect_direction;“支持某个主张”不等于“结果是正向”。</p></div><div class="outcome-groups"><article class="outcome-group outcome-learning"><h3>效果 / 目标结果</h3><ul><li><strong>政策有效性</strong><span class="outcome-states"><span class="dir pos">正向效应 2</span></span></li></ul></article><article class="outcome-group outcome-risk"><h3>风险 / 实施风险</h3><ul><li><strong>实施风险</strong><span class="outcome-states"><span class="dir neg">负向效应 2</span></span></li></ul></article></div></div></div></section><section class="brief-block brief-tribunal" id="brief-tribunal-zh"><header class="brief-block-header"><h2>证据裁决</h2><p>支持、不确定、被反驳与缺失证据分开放置,不把长段落平铺在同一层。</p></header><div class="brief-block-body"><div class="evidence-tribunal" data-visual="evidence-tribunal-grid"><p class="tribunal-summary"><strong>决策:</strong>试点验证 · <strong>置信度:</strong>中</p><div class="tribunal-grid"><article class="tribunal-card supported"><header><span aria-hidden="true">✓</span><h3>可以主张</h3><span class="tribunal-count">4</span></header><ul><li><p>在有边界的客服场景中,AI 辅助可缩短会话处理时间;吞吐量是另一项指标。 — E-001</p><div class="tribunal-evidence-refs"><code>E-001</code></div></li><li><p>ChatGPT 缩短了短篇职业写作任务用时;对客服而言属于间接证据。 — E-002</p><div class="tribunal-evidence-refs"><code>E-002</code></div></li><li><p>在超出能力边界的任务中 AI 可能降低正确率;咨询实验对客服属于间接证据。 — E-003</p><div class="tribunal-evidence-refs"><code>E-003</code></div></li></ul><details class="tribunal-more"><summary>查看其余 1 条</summary><ul><li><p>须单独监测资深高技能客服的质量,不能假定所有岗位群体均受益。 — E-004</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li></ul></details></article><article class="tribunal-card uncertain"><header><span aria-hidden="true">?</span><h3>尚不能主张</h3><span class="tribunal-count">1</span></header><ul><li><p>这组选编证据尚未确立本企业净成本、隐私事件率、亚组服务质量及持续表现 [无直接证据]。</p></li></ul></article><article class="tribunal-card contradicted"><header><span aria-hidden="true">×</span><h3>被反驳的主张</h3><span class="tribunal-count">0</span></header><ul><li>无数据。</li></ul></article><article class="tribunal-card missing"><header><span aria-hidden="true">…</span><h3>缺失证据</h3><span class="tribunal-count">1</span></header><ul><li><p>这组选编证据尚未确立本企业净成本、隐私事件率、亚组服务质量及持续表现 [无直接证据]。</p></li></ul></article></div></div></div></section><section class="brief-block brief-action" id="brief-action-zh"><header class="brief-block-header"><h2>从证据到行动</h2><p>适用性、护栏、停止条件与评价连成一条可执行路径。</p></header><div class="brief-block-body"><div class="evidence-to-action" data-visual="evidence-to-action"><article class="action-node action-evidence"><span>证据</span><p class="action-node-text">在有边界的客服场景中,AI 辅助可缩短会话处理时间;吞吐量是另一项指标。 — E-001</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-applicability"><span>适用性</span><p class="action-node-text">从低风险且知识库覆盖的队列开始;客服审查并编辑每条建议回复。; 按资历和基线技能分层;保留专家判断并记录复核工作量。; 试点前落实客户数据最小化与脱敏、访问和留存限制,并审查供应商数据使用条款。; 陌生、模糊、敏感或高风险请求交由合格人员处理;禁止自动作出承诺。</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-decision"><span>决策</span><p class="action-node-text">试点验证</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-guardrails"><span>护栏</span><p class="action-node-text">从低风险且知识库覆盖的队列开始;客服审查并编辑每条建议回复。</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-stop"><span>停止条件</span><p class="action-node-text">一旦核实隐私泄漏、严重不安全建议或越权客户承诺,立即暂停,调查后再决定是否恢复。; 若盲评质量越过预先约定的非劣界值(包括各资历分层),暂停扩展。</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-evaluation"><span>评价</span><p class="action-node-text">仅在质量非劣、每付薪工时有效解决量提高、净成本可接受且无未解决严重安全事件时考虑扩展;阈值须在本地预先约定。</p></article></div></div></section><section class="brief-block brief-lieflat" id="brief-lieflat-zh"><header class="brief-block-header"><h2>Lieflat 实证手作画廊</h2><p>AI 按数据形状从 Lieflat 目录选型编排;每张图的数字都可溯源到 result.json。</p></header><div class="brief-block-body"><div class="lieflat-gallery-container"><figure class="lieflat-card" data-lieflat data-visual="lieflat-bubble_almanac" data-chart-id="lieflat-bubble-almanac.svg"><h3 class="lieflat-title">发表年份 × 结果维度文献年历</h3><p class="lieflat-sub">气泡面积 ∝ 该格研究数 · 实心圆 = 有显著结果</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 540 300" width="100%" height="100%" role="img" aria-label="发表年份 × 结果维度文献年历" style="background:#FAFAFA;">
|
|
1930
1930
|
<line x1="44" y1="70.0" x2="520" y2="70.0" stroke="#E2E8F0" stroke-width="0.5" class="lf-fade" style="--motion-delay:0ms"/>
|
|
1931
1931
|
<line x1="44" y1="77.0" x2="520" y2="77.0" stroke="#E2E8F0" stroke-width="0.5" class="lf-fade" style="--motion-delay:14ms"/>
|
|
1932
1932
|
<line x1="44" y1="84.0" x2="520" y2="84.0" stroke="#E2E8F0" stroke-width="0.5" class="lf-fade" style="--motion-delay:28ms"/>
|
|
@@ -1955,27 +1955,27 @@ select:focus-visible,
|
|
|
1955
1955
|
<line x1="44" y1="245.0" x2="520" y2="245.0" stroke="#E2E8F0" stroke-width="0.5" class="lf-fade" style="--motion-delay:350ms"/>
|
|
1956
1956
|
<line x1="44" y1="252.0" x2="520" y2="252.0" stroke="#E2E8F0" stroke-width="0.5" class="lf-fade" style="--motion-delay:364ms"/>
|
|
1957
1957
|
<line x1="44" y1="259.0" x2="520" y2="259.0" stroke="#E2E8F0" stroke-width="0.5" class="lf-fade" style="--motion-delay:378ms"/>
|
|
1958
|
-
<text x="150" y="76" fill="#0F172A" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:40ms"
|
|
1959
|
-
<text x="
|
|
1958
|
+
<text x="150" y="76" fill="#0F172A" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:40ms">政策有效性</text>
|
|
1959
|
+
<text x="514.0" y="76" fill="#0F172A" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:140ms">实施风险</text>
|
|
1960
1960
|
<text x="96" y="96.0" fill="#475569" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:80ms">2023</text>
|
|
1961
1961
|
<text x="96" y="140.0" fill="#475569" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:180ms">2025</text>
|
|
1962
1962
|
<text x="96" y="184.0" fill="#475569" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:280ms">2026</text>
|
|
1963
|
-
<circle cx="150.0" cy="92.0" r="3.6" fill="#0F172A" fill-opacity="0.22" stroke="#0F172A" stroke-width="1.2" class="lf-pop" style="--motion-delay:100ms"><title
|
|
1964
|
-
<circle cx="150.0" cy="136.0" r="3.6" fill="#0F172A" fill-opacity="0.22" stroke="#0F172A" stroke-width="1.2" class="lf-pop" style="--motion-delay:112ms"><title
|
|
1965
|
-
<circle cx="520.0" cy="136.0" r="3.6" fill="#0F172A" fill-opacity="0.22" stroke="#0F172A" stroke-width="1.2" class="lf-pop" style="--motion-delay:124ms"><title
|
|
1966
|
-
<circle cx="520.0" cy="180.0" r="3.6" fill="#0F172A" fill-opacity="0.22" stroke="#0F172A" stroke-width="1.2" class="lf-pop" style="--motion-delay:136ms"><title
|
|
1963
|
+
<circle cx="150.0" cy="92.0" r="3.6" fill="#0F172A" fill-opacity="0.22" stroke="#0F172A" stroke-width="1.2" class="lf-pop" style="--motion-delay:100ms"><title>政策有效性 (2023) — N = 1 篇研究, 显著 = 0</title></circle>
|
|
1964
|
+
<circle cx="150.0" cy="136.0" r="3.6" fill="#0F172A" fill-opacity="0.22" stroke="#0F172A" stroke-width="1.2" class="lf-pop" style="--motion-delay:112ms"><title>政策有效性 (2025) — N = 1 篇研究, 显著 = 0</title></circle>
|
|
1965
|
+
<circle cx="520.0" cy="136.0" r="3.6" fill="#0F172A" fill-opacity="0.22" stroke="#0F172A" stroke-width="1.2" class="lf-pop" style="--motion-delay:124ms"><title>实施风险 (2025) — N = 1 篇研究, 显著 = 0</title></circle>
|
|
1966
|
+
<circle cx="520.0" cy="180.0" r="3.6" fill="#0F172A" fill-opacity="0.22" stroke="#0F172A" stroke-width="1.2" class="lf-pop" style="--motion-delay:136ms"><title>实施风险 (2026) — N = 1 篇研究, 显著 = 0</title></circle>
|
|
1967
1967
|
</svg></div><figcaption class="lieflat-caption">仅当证据集携带发表年份与结果维度时绘制。</figcaption><p class="lieflat-src">L9 Bubble Almanac · Evidence.year X Dimension</p></figure><figure class="lieflat-card" data-lieflat data-visual="lieflat-matrix_heat" data-chart-id="lieflat-matrix-heat.svg"><h3 class="lieflat-title">年份 × 结果维度证据密度</h3><p class="lieflat-sub">每格数字 = 该年份该结果维度的证据条数</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 540 200" width="100%" height="100%" role="img" aria-label="年份 × 结果维度证据密度" style="background:#FAFAFA;">
|
|
1968
1968
|
<text x="211.7" y="66" text-anchor="middle" font-size="9" font-weight="800" fill="#475569" class="lf-fade" style="--motion-delay:40ms">2023</text>
|
|
1969
1969
|
<text x="335.0" y="66" text-anchor="middle" font-size="9" font-weight="800" fill="#475569" class="lf-fade" style="--motion-delay:140ms">2025</text>
|
|
1970
1970
|
<text x="458.3" y="66" text-anchor="middle" font-size="9" font-weight="800" fill="#475569" class="lf-fade" style="--motion-delay:240ms">2026</text>
|
|
1971
|
-
<text x="138" y="97.0" text-anchor="end" font-size="8" font-weight="600" fill="#475569" class="lf-fade" style="--motion-delay:60ms"
|
|
1971
|
+
<text x="138" y="97.0" text-anchor="end" font-size="8" font-weight="600" fill="#475569" class="lf-fade" style="--motion-delay:60ms">政策有效性</text>
|
|
1972
1972
|
<rect x="152.0" y="80.0" width="119.3" height="28" rx="4" fill="#0F172A" fill-opacity="0.90" class="lf-pop" style="--motion-delay:0ms"/>
|
|
1973
1973
|
<text x="211.7" y="97.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#0F172A" class="lf-pop" style="--motion-delay:60ms">1</text>
|
|
1974
1974
|
<rect x="275.3" y="80.0" width="119.3" height="28" rx="4" fill="#0F172A" fill-opacity="0.90" class="lf-pop" style="--motion-delay:12ms"/>
|
|
1975
1975
|
<text x="335.0" y="97.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#0F172A" class="lf-pop" style="--motion-delay:72ms">1</text>
|
|
1976
1976
|
<rect x="398.7" y="80.0" width="119.3" height="28" rx="4" fill="#E2E8F0" fill-opacity="0.5" class="lf-fade" style="--motion-delay:24ms"/>
|
|
1977
1977
|
<text x="458.3" y="97.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#0F172A" class="lf-pop" style="--motion-delay:84ms">0</text>
|
|
1978
|
-
<text x="138" y="129.0" text-anchor="end" font-size="8" font-weight="600" fill="#475569" class="lf-fade" style="--motion-delay:160ms"
|
|
1978
|
+
<text x="138" y="129.0" text-anchor="end" font-size="8" font-weight="600" fill="#475569" class="lf-fade" style="--motion-delay:160ms">实施风险</text>
|
|
1979
1979
|
<rect x="152.0" y="112.0" width="119.3" height="28" rx="4" fill="#E2E8F0" fill-opacity="0.5" class="lf-fade" style="--motion-delay:12ms"/>
|
|
1980
1980
|
<text x="211.7" y="129.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#0F172A" class="lf-pop" style="--motion-delay:72ms">0</text>
|
|
1981
1981
|
<rect x="275.3" y="112.0" width="119.3" height="28" rx="4" fill="#0F172A" fill-opacity="0.90" class="lf-pop" style="--motion-delay:24ms"/>
|
|
@@ -2132,7 +2132,7 @@ select:focus-visible,
|
|
|
2132
2132
|
</svg></div><figcaption class="lieflat-caption">让读者一眼看出方法学短板集中在哪些检查项。</figcaption><p class="lieflat-src">L15 Ballot Tally · Methodology.flag Rates</p></figure></div></div></section><section class="brief-block brief-sources" id="brief-sources-zh"><header class="brief-block-header"><h2>关键来源</h2><p>摘要页只列最关键的来源;完整溯源在完整报告中展开。</p></header><div class="brief-block-body"><div class="brief-source-grid"><article class="brief-source"><code>S-001</code><h3><a href="https://academic.oup.com/qje/article/140/2/889/7990658">Generative AI at Work</a></h3><p>T1 DOI 可验证论文 · 2025</p></article><article class="brief-source"><code>S-002</code><h3><a href="https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf">Experimental evidence on the productivity effects of generative artificial intelligence</a></h3><p>T1 DOI 可验证论文 · 2023</p></article><article class="brief-source"><code>S-003</code><h3><a href="https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838">Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality</a></h3><p>T1 DOI 可验证论文 · 2026</p></article></div></div></section></div></div>
|
|
2133
2133
|
<div class="report-page report-page-full" data-report-page="full" hidden>
|
|
2134
2134
|
<div class="full-report-intro"><h2>完整报告</h2><p>结论前置:全部可追溯证据与方法学细节都在这里,关键论证位置穿插有意义的可视化,每个数字都能回查到 result.json。</p></div>
|
|
2135
|
-
<div class="full-report-layout"><aside class="full-report-toc" aria-label="目录"><div class="toc-head"><strong>目录</strong><button type="button" class="toc-collapse" aria-expanded="true" data-label-collapse="收起目录" data-label-expand="展开目录">收起目录</button></div><nav><a href="#full-01-decision" data-toc-target="full-01-decision" data-chapter-key="decision">01 结论、裁决与研究边界</a><a href="#full-02-evidence" data-toc-target="full-02-evidence" data-chapter-key="evidence">02 关键证据与结果分离</a><a href="#full-03-quality" data-toc-target="full-03-quality" data-chapter-key="quality">03 证据可信度、反证与方法审计</a><a href="#full-04-action" data-toc-target="full-04-action" data-chapter-key="action">04
|
|
2135
|
+
<div class="full-report-layout"><aside class="full-report-toc" aria-label="目录"><div class="toc-head"><strong>目录</strong><button type="button" class="toc-collapse" aria-expanded="true" data-label-collapse="收起目录" data-label-expand="展开目录">收起目录</button></div><nav><a href="#full-01-decision" data-toc-target="full-01-decision" data-chapter-key="decision">01 结论、裁决与研究边界</a><a href="#full-02-evidence" data-toc-target="full-02-evidence" data-chapter-key="evidence">02 关键证据与结果分离</a><a href="#full-03-quality" data-toc-target="full-03-quality" data-chapter-key="quality">03 证据可信度、反证与方法审计</a><a href="#full-04-action" data-toc-target="full-04-action" data-chapter-key="action">04 适用范围与政策行动</a><a href="#full-05-evaluation" data-toc-target="full-05-evaluation" data-chapter-key="evaluation">05 试点设计、评估与停止条件</a><a href="#full-06-sources" data-toc-target="full-06-sources" data-chapter-key="sources">06 来源、溯源与附录</a></nav></aside><main class="full-report-content"><section id="full-01-decision" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>01 结论、裁决与研究边界</h2><p class="full-chapter-lead">先明确最终裁决与研究边界,再解释为什么。</p></header><div class="full-chapter-body">
|
|
2136
2136
|
<div class="decision-hero pilot" data-visual="decision-hero">
|
|
2137
2137
|
<div class="hero-decision">
|
|
2138
2138
|
<span class="eyebrow">建议决策</span>
|
|
@@ -2147,19 +2147,19 @@ select:focus-visible,
|
|
|
2147
2147
|
<article class="hero-insight next"><span>下一步</span><p class="hero-insight-text">在已批准知识库上开展人工监督试点,每条回复均经坐席复核,并跟踪升级率与纠错率。</p></article>
|
|
2148
2148
|
</div>
|
|
2149
2149
|
<p class="hero-provenance"><span>证据 / 来源</span> · 4 / 3</p>
|
|
2150
|
-
</div><div class="scope-grid"><article class="scope-card"><h3>研究问题</h3><p class="scope-text">企业客服团队是否应引入生成式 AI 助手?</p></article><article class="scope-card"><h3>AI 干预</h3><p class="scope-text">政策名称:人工监督的客服助手;政策类型:机构改革;作用机制:已批准知识库 + 坐席复核</p></article><article class="scope-card"><h3>比较条件</h3><p class="scope-text">不提供生成式 AI 建议的现有客服流程。</p></article><article class="scope-card"><h3>结果构念</h3><p class="scope-text">主要结果:Policy Effectiveness、Implementation Risk;次要结果:Cost Effectiveness、equity、feasibility</p></article><article class="scope-card"><h3>研究范围</h3><p class="scope-text">时间范围:2023–2026;Evidence Types:准实验、随机对照试验</p></article><article class="scope-card"><h3>决策成功条件</h3><p class="scope-text">提高每付薪工时的有效解决量,同时维持服务质量、隐私和员工自主判断。</p></article></div></div></section><section id="full-02-evidence" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>02 关键证据与结果分离</h2><p class="full-chapter-lead">把任务表现、真实学习、保持与风险放在同一证据地图中,但不混为一谈。</p></header><div class="full-chapter-body"><div class="retrieval-grid"><article><h3>纳入标准</h3><ul></ul></article><article><h3>排除标准</h3><ul></ul></article></div><div class="retrieval-coverage"><h3>证据来源覆盖</h3><p><code>S-001</code> <code>S-002</code> <code>S-003</code></p><p class="retrieval-note">当前报告只展示 result 中真实存在的检索与来源信息;没有流程计数时不伪造 PRISMA / funnel 数字。</p></div><div class="outcome-separation" data-visual="outcome-separation"><div class="visual-heading"><h3>结果分离 · 任务表现 ≠ 学习效果</h3><p>将不同结果类型分开裁决,避免把训练时更快、更高分直接等同于真正学会。</p><p class="semantic-note">效应方向来自 evidence.effect_direction;“支持某个主张”不等于“结果是正向”。</p></div><div class="outcome-groups"><article class="outcome-group outcome-other"><h3>其他结果</h3><ul><li><strong>Policy Effectiveness</strong><span class="outcome-states"><span class="dir pos">正向效应 2</span></span></li><li><strong>Implementation Risk</strong><span class="outcome-states"><span class="dir neg">负向效应 2</span></span></li></ul></article></div></div><div class='table-wrap outcome-table'><table class='data-table'><thead><tr><th>结果类型</th><th>正向效应</th><th>负向效应</th><th>零效应</th><th>证据</th></tr></thead><tbody><tr><td><strong>Policy Effectiveness</strong><span class='raw-tag' title='原始标识'>policy_effectiveness</span></td><td class='num'>2</td><td class='num'>0</td><td class='num'>0</td><td><code>E-001</code> <code>E-002</code> </td></tr><tr><td><strong>Implementation Risk</strong><span class='raw-tag' title='原始标识'>implementation_risk</span></td><td class='num'>0</td><td class='num'>2</td><td class='num'>0</td><td><code>E-003</code> <code>E-004</code> </td></tr></tbody></table></div><div class='matrix-controls'><div class='matrix-tools'><input id='matrix-search-full-zh' type='search' placeholder='搜索证据…' aria-label='筛选 / 搜索证据'><select id='matrix-direction-full-zh' aria-label='按效应方向筛选'><option value=''>全部效应</option><option value='positive'>正向效应</option><option value='negative'>负向效应</option><option value='null'>零效应</option></select><select id='matrix-outcome-full-zh' aria-label='按结果类型筛选'><option value=''>全部结果</option><option value='implementation_risk'>Implementation Risk</option><option value='policy_effectiveness'>Policy Effectiveness</option></select></div></div><div class='table-wrap matrix-wrap'><table id='evidence-matrix-full-zh' class='data-table evidence-matrix'><thead><tr><th>ID</th><th>结果</th><th>效应</th><th>质量</th><th>主张</th><th>来源</th></tr></thead><tbody><tr data-effect="positive" data-direction="positive" data-outcome="policy_effectiveness" data-search="e-001 st-001 sample-support-all generative ai at work 在有边界的客服场景中,ai 辅助可缩短会话处理时间;吞吐量是另一项指标。 客服人员 positive s-001"><td><code>E-001</code></td><td><strong>Policy Effectiveness</strong></td><td><span class="dir pos">正向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">9</span><span class="quality-meter" aria-hidden="true"><i style="width:90%"></i></span></div></td><td class="claim-cell"><p>在有边界的客服场景中,AI 辅助可缩短会话处理时间;吞吐量是另一项指标。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>ST-001</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SAMPLE-support-all</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Generative AI at Work</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Generative AI at Work</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>准实验</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>客服人员</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>5172</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>会话处理时间;另列吞吐量摘要:每小时解决问题数约增加 15%。</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>正向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>单一企业、单一工具、非随机分批上线;因果解释依赖识别假设。</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=1 · D2 样本质量=2 · D3 测量效度=2 · D4 时间强度=2 · D5 直接性=2</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>9.0</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>在有边界的客服场景中,AI 辅助可缩短会话处理时间;吞吐量是另一项指标。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://academic.oup.com/qje/article/140/2/889/7990658</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://academic.oup.com/qje/article/140/2/889/7990658">https://academic.oup.com/qje/article/140/2/889/7990658</a></dd></div></dl></details></td><td><a class="source-link" href="https://academic.oup.com/qje/article/140/2/889/7990658" title="Generative AI at Work"><code>S-001</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="policy_effectiveness" data-search="e-002 st-002 sample-writing experimental evidence on the productivity effects of generative artificial intelligence chatgpt 缩短了短篇职业写作任务用时;对客服而言属于间接证据。 受过大学教育的职场人士 positive s-002"><td><code>E-002</code></td><td><strong>Policy Effectiveness</strong></td><td><span class="dir pos">正向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>ChatGPT 缩短了短篇职业写作任务用时;对客服而言属于间接证据。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>ST-002</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SAMPLE-writing</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Experimental evidence on the productivity effects of generative artificial intelligence</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Experimental evidence on the productivity effects of generative artificial intelligence</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>随机对照试验</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>受过大学教育的职场人士</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>453</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>自报任务用时约减少 40%;评定写作质量约提高 18% 是另一项指标。</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>正向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>间接证据,不可直接推广:短时激励写作任务不要求精确事实或客户特定情境。</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=2 · D2 样本质量=2 · D3 测量效度=1 · D4 时间强度=1 · D5 直接性=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>ChatGPT 缩短了短篇职业写作任务用时;对客服而言属于间接证据。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf">https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf</a></dd></div></dl></details></td><td><a class="source-link" href="https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf" title="Experimental evidence on the productivity effects of generative artificial intelligence"><code>S-002</code></a></td></tr><tr data-effect="negative" data-direction="negative" data-outcome="implementation_risk" data-search="e-003 st-003 sample-consulting-outside navigating the jagged technological frontier: field experimental evidence of the effects of artificial intelligence on knowledge worker productivity and quality 在超出能力边界的任务中 ai 可能降低正确率;咨询实验对客服属于间接证据。 参与超出能力边界实验的 bcg 顾问 negative s-003"><td><code>E-003</code></td><td><strong>Implementation Risk</strong></td><td><span class="dir neg">负向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>在超出能力边界的任务中 AI 可能降低正确率;咨询实验对客服属于间接证据。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>ST-003</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SAMPLE-consulting-outside</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2026</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>随机对照试验</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>参与超出能力边界实验的 BCG 顾问</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>373</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>商业建议正确率:合并 AI 组约低 19 个百分点;表 7 为 373 人,不是 758 人。</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>负向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>间接证据,不可直接推广:咨询顾问处理实验商业案例,并非真实客服工单。</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=2 · D2 样本质量=2 · D3 测量效度=2 · D4 时间强度=1 · D5 直接性=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>在超出能力边界的任务中 AI 可能降低正确率;咨询实验对客服属于间接证据。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838">https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838</a></dd></div></dl></details></td><td><a class="source-link" href="https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838" title="Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality"><code>S-003</code></a></td></tr><tr data-effect="negative" data-direction="negative" data-outcome="implementation_risk" data-search="e-004 st-001 sample-support-all generative ai at work 须单独监测资深高技能客服的质量,不能假定所有岗位群体均受益。 资深高技能客服人员 negative s-001"><td><code>E-004</code></td><td><strong>Implementation Risk</strong></td><td><span class="dir neg">负向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>须单独监测资深高技能客服的质量,不能假定所有岗位群体均受益。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>ST-001</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SAMPLE-support-all</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Generative AI at Work</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Generative AI at Work</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>准实验</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>资深高技能客服人员</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>资历最深、技能最高的客服群体出现小幅质量下降。</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>负向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>与 E-001 属于同一研究;未提取该亚组样本量,不是独立重复验证。</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=1 · D2 样本质量=1 · D3 测量效度=2 · D4 时间强度=2 · D5 直接性=2</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>须单独监测资深高技能客服的质量,不能假定所有岗位群体均受益。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://academic.oup.com/qje/article/140/2/889/7990658</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://academic.oup.com/qje/article/140/2/889/7990658">https://academic.oup.com/qje/article/140/2/889/7990658</a></dd></div></dl></details></td><td><a class="source-link" href="https://academic.oup.com/qje/article/140/2/889/7990658" title="Generative AI at Work"><code>S-001</code></a></td></tr></tbody></table></div></div></section><section id="full-03-quality" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>03 证据可信度、反证与方法审计</h2><p class="full-chapter-lead">检查证据为什么可信、哪里冲突,以及哪些结论必须降级。</p></header><div class="full-chapter-body"><section class="method-review-group"><header class="method-review-title"><h3>审查目标:overall</h3><span class="method-verdict">关注</span></header><div class="method-audit-grid"><article class="method-audit-item method-met"><div class="method-audit-head"><strong>evidence level</strong><span class="method-status">通过</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>causal identification</strong><span class="method-status">部分</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>external validity</strong><span class="method-status">部分</span></div><p></p></article><article class="method-audit-item method-missing"><div class="method-audit-head"><strong>cost evidence</strong><span class="method-status">缺失</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>stakeholder representation</strong><span class="method-status">部分</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>implementation evidence</strong><span class="method-status">部分</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>equity analysis</strong><span class="method-status">部分</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>effect size reported</strong><span class="method-status">部分</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>uncertainty quantified</strong><span class="method-status">部分</span></div><p></p></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>comparator clarity</strong><span class="method-status">通过</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>publication bias risk</strong><span class="method-status">部分</span></div><p></p></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>generalizability claims</strong><span class="method-status">通过</span></div><p></p></article></div><div class="method-guard-wrap"><strong>任务 vs 学习护栏:</strong><p class="method-guard">教学结果不适用;生产率并非学生学习效果主张。</p></div></section><p>无数据。</p><div class="evidence-tribunal" data-visual="evidence-tribunal-grid"><p class="tribunal-summary"><strong>决策:</strong>试点验证 · <strong>置信度:</strong>中</p><div class="tribunal-grid"><article class="tribunal-card supported"><header><span aria-hidden="true">✓</span><h3>可以主张</h3><span class="tribunal-count">4</span></header><ul><li><p>在有边界的客服场景中,AI 辅助可缩短会话处理时间;吞吐量是另一项指标。 — E-001</p><div class="tribunal-evidence-refs"><code>E-001</code></div></li><li><p>ChatGPT 缩短了短篇职业写作任务用时;对客服而言属于间接证据。 — E-002</p><div class="tribunal-evidence-refs"><code>E-002</code></div></li><li><p>在超出能力边界的任务中 AI 可能降低正确率;咨询实验对客服属于间接证据。 — E-003</p><div class="tribunal-evidence-refs"><code>E-003</code></div></li><li><p>须单独监测资深高技能客服的质量,不能假定所有岗位群体均受益。 — E-004</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li></ul></article><article class="tribunal-card uncertain"><header><span aria-hidden="true">?</span><h3>尚不能主张</h3><span class="tribunal-count">1</span></header><ul><li><p>这组选编证据尚未确立本企业净成本、隐私事件率、亚组服务质量及持续表现 [无直接证据]。</p></li></ul></article><article class="tribunal-card contradicted"><header><span aria-hidden="true">×</span><h3>被反驳的主张</h3><span class="tribunal-count">0</span></header><ul><li>无数据。</li></ul></article><article class="tribunal-card missing"><header><span aria-hidden="true">…</span><h3>缺失证据</h3><span class="tribunal-count">1</span></header><ul><li><p>这组选编证据尚未确立本企业净成本、隐私事件率、亚组服务质量及持续表现 [无直接证据]。</p></li></ul></article></div><details class="supporting-visual"><summary>EvidenceFlow 协议</summary><svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="EvidenceFlow 协议流程"><title>EvidenceFlow 协议流程</title><desc>从问题框架、检索、抓取验证、证据抽取、反方质疑、方法审计、裁决到适用性与干预评价的完整流程。</desc><defs><marker id="arr-zh-full-workflow" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">EvidenceFlow 协议</text><rect x="24" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="56.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="56.0" y="126.0">问题框架</tspan></text><line x1="88" y1="126" x2="100" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="100" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="132.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="132.0" y="126.0">检索</tspan></text><line x1="164" y1="126" x2="176" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="176" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="208.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="208.0" y="126.0">抓取</tspan><tspan x="208.0" y="140.0">验证</tspan></text><line x1="240" y1="126" x2="252" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="252" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="284.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="284.0" y="126.0">证据抽取</tspan></text><line x1="316" y1="126" x2="328" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="328" y="100" width="64" height="52" rx="8" fill="#5E8A6A"/><text x="360.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="360.0" y="126.0">反方质疑</tspan></text><line x1="392" y1="126" x2="404" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="404" y="100" width="64" height="52" rx="8" fill="#5E8A6A"/><text x="436.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="436.0" y="126.0">方法</tspan><tspan x="436.0" y="140.0">审计</tspan></text><line x1="468" y1="126" x2="480" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="480" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="512.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="512.0" y="126.0">裁决</tspan></text><line x1="544" y1="126" x2="556" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="556" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="588.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="588.0" y="126.0">适用性</tspan></text><line x1="620" y1="126" x2="632" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="632" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="664.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="664.0" y="126.0">干预</tspan><tspan x="664.0" y="140.0">评价</tspan></text></svg></details><details class="supporting-visual"><summary>裁决信息图</summary><svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="证据裁决信息图"><title>证据裁决信息图</title><desc>可以主张与不可主张的证据 ID 与建议决策徽章;完整主张文本见下方裁决卡片。</desc><defs><marker id="arr-zh-full-tribunal" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">证据裁决</text><text x="24" y="180" font-size="13" font-weight="700" fill="#3A3833">冲突来源</text><text x="24" y="202" font-size="11" fill="#8A867E">详见下方裁决卡片</text><rect x="24" y="216" width="180" height="28" rx="14" fill="#C99A4A"/><text x="114" y="235" text-anchor="middle" font-size="13" font-weight="700" fill="#fff">试点验证</text><text x="250" y="80" font-size="13" font-weight="700" fill="#5E8A6A">可以主张 (4)</text><circle cx="254" cy="100" r="3" fill="#5E8A6A"/><text x="266" y="105" font-size="11" fill="#3A3833">E-001</text><circle cx="254" cy="122" r="3" fill="#5E8A6A"/><text x="266" y="127" font-size="11" fill="#3A3833">E-002</text><circle cx="254" cy="144" r="3" fill="#5E8A6A"/><text x="266" y="149" font-size="11" fill="#3A3833">E-003</text><circle cx="254" cy="166" r="3" fill="#5E8A6A"/><text x="266" y="171" font-size="11" fill="#3A3833">E-004</text><text x="490" y="80" font-size="13" font-weight="700" fill="#A85B53">不可主张 (1)</text><circle cx="494" cy="100" r="3" fill="#A85B53"/><text x="506" y="105" font-size="11" fill="#3A3833">(无)</text></svg></details></div><div class="trace-chain" data-visual="trace-chain"><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-001</strong><p class="trace-claim-text">在有边界的客服场景中,AI 辅助可缩短会话处理时间;吞吐量是另一项指标。</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-001</code><span>Policy Effectiveness</span><span class="dir pos">正向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 9.0</span><span class="trace-arrow">→</span><a href="https://academic.oup.com/qje/article/140/2/889/7990658"><code>S-001</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-002</strong><p class="trace-claim-text">ChatGPT 缩短了短篇职业写作任务用时;对客服而言属于间接证据。</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-002</code><span>Policy Effectiveness</span><span class="dir pos">正向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf"><code>S-002</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-003</strong><p class="trace-claim-text">在超出能力边界的任务中 AI 可能降低正确率;咨询实验对客服属于间接证据。</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-003</code><span>Implementation Risk</span><span class="dir neg">负向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838"><code>S-003</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-004</strong><p class="trace-claim-text">须单独监测资深高技能客服的质量,不能假定所有岗位群体均受益。</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-004</code><span>Implementation Risk</span><span class="dir neg">负向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://academic.oup.com/qje/article/140/2/889/7990658"><code>S-001</code></a></div></div></article></div><div id="chart-trace-zh" class="chart-mount" aria-label="主张-证据追溯"></div><p class="chart-interpretation"><strong>这意味着什么:</strong>每个重要主张都必须能追到 Evidence ID 和原始来源。</p></div></section><section id="full-04-action" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>04 适用范围与教学行动</h2><p class="full-chapter-lead">把可外推范围、护栏和教学动作连接到具体证据。</p></header><div class="full-chapter-body"><p><strong>目标人群:</strong>企业客服人员,按资历和基线技能分层。</p>
|
|
2150
|
+
</div><div class="scope-grid"><article class="scope-card"><h3>研究问题</h3><p class="scope-text">企业客服团队是否应引入生成式 AI 助手?</p></article><article class="scope-card"><h3>干预方案</h3><p class="scope-text">政策名称:人工监督的客服助手;政策类型:机构改革;作用机制:已批准知识库 + 坐席复核</p></article><article class="scope-card"><h3>比较条件</h3><p class="scope-text">不提供生成式 AI 建议的现有客服流程。</p></article><article class="scope-card"><h3>结果构念</h3><p class="scope-text">主要结果:政策有效性、实施风险;次要结果:成本效益、公平性、可行性</p></article><article class="scope-card"><h3>研究范围</h3><p class="scope-text">时间范围:2023–2026;Evidence Types:准实验、随机对照试验</p></article><article class="scope-card"><h3>决策成功条件</h3><p class="scope-text">提高每付薪工时的有效解决量,同时维持服务质量、隐私和员工自主判断。</p></article></div></div></section><section id="full-02-evidence" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>02 关键证据与结果分离</h2><p class="full-chapter-lead">把过程产出、真实效果、持续性与风险放在同一证据地图中,但不混为一谈。</p></header><div class="full-chapter-body"><div class="retrieval-grid"><article><h3>纳入标准</h3><ul></ul></article><article><h3>排除标准</h3><ul></ul></article></div><div class="retrieval-coverage"><h3>证据来源覆盖</h3><p><code>S-001</code> <code>S-002</code> <code>S-003</code></p><p class="retrieval-note">当前报告只展示 result 中真实存在的检索与来源信息;没有流程计数时不伪造 PRISMA / funnel 数字。</p></div><div class="outcome-separation" data-visual="outcome-separation"><div class="visual-heading"><h3>结果分离 · 过程产出 ≠ 政策效果</h3><p>将不同结果类型分开裁决,避免把试点期产出或效率提升直接等同于真实政策效果。</p><p class="semantic-note">效应方向来自 evidence.effect_direction;“支持某个主张”不等于“结果是正向”。</p></div><div class="outcome-groups"><article class="outcome-group outcome-learning"><h3>效果 / 目标结果</h3><ul><li><strong>政策有效性</strong><span class="outcome-states"><span class="dir pos">正向效应 2</span></span></li></ul></article><article class="outcome-group outcome-risk"><h3>风险 / 实施风险</h3><ul><li><strong>实施风险</strong><span class="outcome-states"><span class="dir neg">负向效应 2</span></span></li></ul></article></div></div><div class='table-wrap outcome-table'><table class='data-table'><thead><tr><th>结果类型</th><th>正向效应</th><th>负向效应</th><th>零效应</th><th>证据</th></tr></thead><tbody><tr><td><strong>政策有效性</strong><span class='raw-tag' title='原始标识'>policy_effectiveness</span></td><td class='num'>2</td><td class='num'>0</td><td class='num'>0</td><td><code>E-001</code> <code>E-002</code> </td></tr><tr><td><strong>实施风险</strong><span class='raw-tag' title='原始标识'>implementation_risk</span></td><td class='num'>0</td><td class='num'>2</td><td class='num'>0</td><td><code>E-003</code> <code>E-004</code> </td></tr></tbody></table></div><div class='matrix-controls'><div class='matrix-tools'><input id='matrix-search-full-zh' type='search' placeholder='搜索证据…' aria-label='筛选 / 搜索证据'><select id='matrix-direction-full-zh' aria-label='按效应方向筛选'><option value=''>全部效应</option><option value='positive'>正向效应</option><option value='negative'>负向效应</option><option value='null'>零效应</option></select><select id='matrix-outcome-full-zh' aria-label='按结果类型筛选'><option value=''>全部结果</option><option value='implementation_risk'>实施风险</option><option value='policy_effectiveness'>政策有效性</option></select></div></div><div class='table-wrap matrix-wrap'><table id='evidence-matrix-full-zh' class='data-table evidence-matrix'><thead><tr><th>ID</th><th>结果</th><th>效应</th><th>质量</th><th>主张</th><th>来源</th></tr></thead><tbody><tr data-effect="positive" data-direction="positive" data-outcome="policy_effectiveness" data-search="e-001 st-001 sample-support-all generative ai at work 在有边界的客服场景中,ai 辅助可缩短会话处理时间;吞吐量是另一项指标。 客服人员 positive s-001"><td><code>E-001</code></td><td><strong>政策有效性</strong></td><td><span class="dir pos">正向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">9</span><span class="quality-meter" aria-hidden="true"><i style="width:90%"></i></span></div></td><td class="claim-cell"><p>在有边界的客服场景中,AI 辅助可缩短会话处理时间;吞吐量是另一项指标。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>ST-001</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SAMPLE-support-all</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Generative AI at Work</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Generative AI at Work</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>准实验</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>客服人员</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>5172</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>会话处理时间;另列吞吐量摘要:每小时解决问题数约增加 15%。</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>正向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>单一企业、单一工具、非随机分批上线;因果解释依赖识别假设。</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=1 · D2 样本质量=2 · D3 测量效度=2 · D4 时间强度=2 · D5 直接性=2</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>9.0</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>在有边界的客服场景中,AI 辅助可缩短会话处理时间;吞吐量是另一项指标。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://academic.oup.com/qje/article/140/2/889/7990658</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://academic.oup.com/qje/article/140/2/889/7990658">https://academic.oup.com/qje/article/140/2/889/7990658</a></dd></div></dl></details></td><td><a class="source-link" href="https://academic.oup.com/qje/article/140/2/889/7990658" title="Generative AI at Work"><code>S-001</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="policy_effectiveness" data-search="e-002 st-002 sample-writing experimental evidence on the productivity effects of generative artificial intelligence chatgpt 缩短了短篇职业写作任务用时;对客服而言属于间接证据。 受过大学教育的职场人士 positive s-002"><td><code>E-002</code></td><td><strong>政策有效性</strong></td><td><span class="dir pos">正向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>ChatGPT 缩短了短篇职业写作任务用时;对客服而言属于间接证据。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>ST-002</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SAMPLE-writing</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Experimental evidence on the productivity effects of generative artificial intelligence</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Experimental evidence on the productivity effects of generative artificial intelligence</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>随机对照试验</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>受过大学教育的职场人士</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>453</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>自报任务用时约减少 40%;评定写作质量约提高 18% 是另一项指标。</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>正向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>间接证据,不可直接推广:短时激励写作任务不要求精确事实或客户特定情境。</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=2 · D2 样本质量=2 · D3 测量效度=1 · D4 时间强度=1 · D5 直接性=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>ChatGPT 缩短了短篇职业写作任务用时;对客服而言属于间接证据。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf">https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf</a></dd></div></dl></details></td><td><a class="source-link" href="https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf" title="Experimental evidence on the productivity effects of generative artificial intelligence"><code>S-002</code></a></td></tr><tr data-effect="negative" data-direction="negative" data-outcome="implementation_risk" data-search="e-003 st-003 sample-consulting-outside navigating the jagged technological frontier: field experimental evidence of the effects of artificial intelligence on knowledge worker productivity and quality 在超出能力边界的任务中 ai 可能降低正确率;咨询实验对客服属于间接证据。 参与超出能力边界实验的 bcg 顾问 negative s-003"><td><code>E-003</code></td><td><strong>实施风险</strong></td><td><span class="dir neg">负向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>在超出能力边界的任务中 AI 可能降低正确率;咨询实验对客服属于间接证据。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>ST-003</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SAMPLE-consulting-outside</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2026</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>随机对照试验</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>参与超出能力边界实验的 BCG 顾问</dd></div><div class="evidence-detail-row"><dt>样本量</dt><dd>373</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>商业建议正确率:合并 AI 组约低 19 个百分点;表 7 为 373 人,不是 758 人。</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>负向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>间接证据,不可直接推广:咨询顾问处理实验商业案例,并非真实客服工单。</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=2 · D2 样本质量=2 · D3 测量效度=2 · D4 时间强度=1 · D5 直接性=1</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>在超出能力边界的任务中 AI 可能降低正确率;咨询实验对客服属于间接证据。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838">https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838</a></dd></div></dl></details></td><td><a class="source-link" href="https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838" title="Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality"><code>S-003</code></a></td></tr><tr data-effect="negative" data-direction="negative" data-outcome="implementation_risk" data-search="e-004 st-001 sample-support-all generative ai at work 须单独监测资深高技能客服的质量,不能假定所有岗位群体均受益。 资深高技能客服人员 negative s-001"><td><code>E-004</code></td><td><strong>实施风险</strong></td><td><span class="dir neg">负向效应</span><span class="relation-note">与主张关系: 支持</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>须单独监测资深高技能客服的质量,不能假定所有岗位群体均受益。</p><details class="matrix-row-detail"><summary>查看完整证据</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>研究 ID</dt><dd>ST-001</dd></div><div class="evidence-detail-row"><dt>样本 ID</dt><dd>SAMPLE-support-all</dd></div><div class="evidence-detail-row"><dt>研究标题</dt><dd>Generative AI at Work</dd></div><div class="evidence-detail-row"><dt>来源标题</dt><dd>Generative AI at Work</dd></div><div class="evidence-detail-row"><dt>年份</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>研究设计</dt><dd>准实验</dd></div><div class="evidence-detail-row"><dt>研究人群</dt><dd>资深高技能客服人员</dd></div><div class="evidence-detail-row"><dt>结果测量</dt><dd>资历最深、技能最高的客服群体出现小幅质量下降。</dd></div><div class="evidence-detail-row"><dt>效应方向</dt><dd>负向效应</dd></div><div class="evidence-detail-row"><dt>与主张关系</dt><dd>支持</dd></div><div class="evidence-detail-row"><dt>局限</dt><dd>与 E-001 属于同一研究;未提取该亚组样本量,不是独立重复验证。</dd></div><div class="evidence-detail-row"><dt>质量维度</dt><dd>D1 研究设计=1 · D2 样本质量=1 · D3 测量效度=2 · D4 时间强度=2 · D5 直接性=2</dd></div><div class="evidence-detail-row"><dt>质量分</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>证据状态</dt><dd>已支持</dd></div><div class="evidence-detail-row"><dt>完整主张</dt><dd>须单独监测资深高技能客服的质量,不能假定所有岗位群体均受益。</dd></div><div class="evidence-detail-row"><dt>来源位置</dt><dd>https://academic.oup.com/qje/article/140/2/889/7990658</dd></div><div class="evidence-detail-row"><dt>可验证链接</dt><dd><a href="https://academic.oup.com/qje/article/140/2/889/7990658">https://academic.oup.com/qje/article/140/2/889/7990658</a></dd></div></dl></details></td><td><a class="source-link" href="https://academic.oup.com/qje/article/140/2/889/7990658" title="Generative AI at Work"><code>S-001</code></a></td></tr></tbody></table></div></div></section><section id="full-03-quality" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>03 证据可信度、反证与方法审计</h2><p class="full-chapter-lead">检查证据为什么可信、哪里冲突,以及哪些结论必须降级。</p></header><div class="full-chapter-body"><section class="method-review-group"><header class="method-review-title"><h3>审查目标:overall</h3><span class="method-verdict">关注</span></header><div class="method-audit-grid"><article class="method-audit-item method-met"><div class="method-audit-head"><strong>证据等级</strong><span class="method-status">通过</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>因果识别</strong><span class="method-status">部分</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>外部效度</strong><span class="method-status">部分</span></div><p></p></article><article class="method-audit-item method-missing"><div class="method-audit-head"><strong>成本证据</strong><span class="method-status">缺失</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>利益相关者代表</strong><span class="method-status">部分</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>实施证据</strong><span class="method-status">部分</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>公平性分析</strong><span class="method-status">部分</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>效应量报告</strong><span class="method-status">部分</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>不确定性量化</strong><span class="method-status">部分</span></div><p></p></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>对照与反事实清晰</strong><span class="method-status">通过</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>发表偏倚风险</strong><span class="method-status">部分</span></div><p></p></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>可推广性声明</strong><span class="method-status">通过</span></div><p></p></article></div><div class="method-guard-wrap"><strong>产出 vs 效果护栏:</strong><p class="method-guard">教学结果不适用;生产率并非学生学习效果主张。</p></div></section><p>无数据。</p><div class="evidence-tribunal" data-visual="evidence-tribunal-grid"><p class="tribunal-summary"><strong>决策:</strong>试点验证 · <strong>置信度:</strong>中</p><div class="tribunal-grid"><article class="tribunal-card supported"><header><span aria-hidden="true">✓</span><h3>可以主张</h3><span class="tribunal-count">4</span></header><ul><li><p>在有边界的客服场景中,AI 辅助可缩短会话处理时间;吞吐量是另一项指标。 — E-001</p><div class="tribunal-evidence-refs"><code>E-001</code></div></li><li><p>ChatGPT 缩短了短篇职业写作任务用时;对客服而言属于间接证据。 — E-002</p><div class="tribunal-evidence-refs"><code>E-002</code></div></li><li><p>在超出能力边界的任务中 AI 可能降低正确率;咨询实验对客服属于间接证据。 — E-003</p><div class="tribunal-evidence-refs"><code>E-003</code></div></li><li><p>须单独监测资深高技能客服的质量,不能假定所有岗位群体均受益。 — E-004</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li></ul></article><article class="tribunal-card uncertain"><header><span aria-hidden="true">?</span><h3>尚不能主张</h3><span class="tribunal-count">1</span></header><ul><li><p>这组选编证据尚未确立本企业净成本、隐私事件率、亚组服务质量及持续表现 [无直接证据]。</p></li></ul></article><article class="tribunal-card contradicted"><header><span aria-hidden="true">×</span><h3>被反驳的主张</h3><span class="tribunal-count">0</span></header><ul><li>无数据。</li></ul></article><article class="tribunal-card missing"><header><span aria-hidden="true">…</span><h3>缺失证据</h3><span class="tribunal-count">1</span></header><ul><li><p>这组选编证据尚未确立本企业净成本、隐私事件率、亚组服务质量及持续表现 [无直接证据]。</p></li></ul></article></div><details class="supporting-visual"><summary>EvidenceFlow 协议</summary><svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="EvidenceFlow 协议流程"><title>EvidenceFlow 协议流程</title><desc>从问题框架、检索、抓取验证、证据抽取、反方质疑、方法审计、裁决到适用性与干预评价的完整流程。</desc><defs><marker id="arr-zh-full-workflow" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">EvidenceFlow 协议</text><rect x="24" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="56.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="56.0" y="126.0">问题框架</tspan></text><line x1="88" y1="126" x2="100" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="100" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="132.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="132.0" y="126.0">检索</tspan></text><line x1="164" y1="126" x2="176" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="176" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="208.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="208.0" y="126.0">抓取</tspan><tspan x="208.0" y="140.0">验证</tspan></text><line x1="240" y1="126" x2="252" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="252" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="284.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="284.0" y="126.0">证据抽取</tspan></text><line x1="316" y1="126" x2="328" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="328" y="100" width="64" height="52" rx="8" fill="#5E8A6A"/><text x="360.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="360.0" y="126.0">反方质疑</tspan></text><line x1="392" y1="126" x2="404" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="404" y="100" width="64" height="52" rx="8" fill="#5E8A6A"/><text x="436.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="436.0" y="126.0">方法</tspan><tspan x="436.0" y="140.0">审计</tspan></text><line x1="468" y1="126" x2="480" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="480" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="512.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="512.0" y="126.0">裁决</tspan></text><line x1="544" y1="126" x2="556" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="556" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="588.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="588.0" y="126.0">适用性</tspan></text><line x1="620" y1="126" x2="632" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-workflow)"/><rect x="632" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="664.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="664.0" y="126.0">干预</tspan><tspan x="664.0" y="140.0">评价</tspan></text></svg></details><details class="supporting-visual"><summary>裁决信息图</summary><svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="证据裁决信息图"><title>证据裁决信息图</title><desc>可以主张与不可主张的证据 ID 与建议决策徽章;完整主张文本见下方裁决卡片。</desc><defs><marker id="arr-zh-full-tribunal" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">证据裁决</text><text x="24" y="180" font-size="13" font-weight="700" fill="#3A3833">冲突来源</text><text x="24" y="202" font-size="11" fill="#8A867E">详见下方裁决卡片</text><rect x="24" y="216" width="180" height="28" rx="14" fill="#C99A4A"/><text x="114" y="235" text-anchor="middle" font-size="13" font-weight="700" fill="#fff">试点验证</text><text x="250" y="80" font-size="13" font-weight="700" fill="#5E8A6A">可以主张 (4)</text><circle cx="254" cy="100" r="3" fill="#5E8A6A"/><text x="266" y="105" font-size="11" fill="#3A3833">E-001</text><circle cx="254" cy="122" r="3" fill="#5E8A6A"/><text x="266" y="127" font-size="11" fill="#3A3833">E-002</text><circle cx="254" cy="144" r="3" fill="#5E8A6A"/><text x="266" y="149" font-size="11" fill="#3A3833">E-003</text><circle cx="254" cy="166" r="3" fill="#5E8A6A"/><text x="266" y="171" font-size="11" fill="#3A3833">E-004</text><text x="490" y="80" font-size="13" font-weight="700" fill="#A85B53">不可主张 (1)</text><circle cx="494" cy="100" r="3" fill="#A85B53"/><text x="506" y="105" font-size="11" fill="#3A3833">(无)</text></svg></details></div><div class="trace-chain" data-visual="trace-chain"><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-001</strong><p class="trace-claim-text">在有边界的客服场景中,AI 辅助可缩短会话处理时间;吞吐量是另一项指标。</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-001</code><span>政策有效性</span><span class="dir pos">正向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 9.0</span><span class="trace-arrow">→</span><a href="https://academic.oup.com/qje/article/140/2/889/7990658"><code>S-001</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-002</strong><p class="trace-claim-text">ChatGPT 缩短了短篇职业写作任务用时;对客服而言属于间接证据。</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-002</code><span>政策有效性</span><span class="dir pos">正向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf"><code>S-002</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-003</strong><p class="trace-claim-text">在超出能力边界的任务中 AI 可能降低正确率;咨询实验对客服属于间接证据。</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-003</code><span>实施风险</span><span class="dir neg">负向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838"><code>S-003</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-004</strong><p class="trace-claim-text">须单独监测资深高技能客服的质量,不能假定所有岗位群体均受益。</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-004</code><span>实施风险</span><span class="dir neg">负向效应</span><span class="trace-relation">主张关系: 支持</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://academic.oup.com/qje/article/140/2/889/7990658"><code>S-001</code></a></div></div></article></div><div id="chart-trace-zh" class="chart-mount" aria-label="主张-证据追溯"></div><p class="chart-interpretation"><strong>这意味着什么:</strong>每个重要主张都必须能追到 Evidence ID 和原始来源。</p></div></section><section id="full-04-action" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>04 适用范围与政策行动</h2><p class="full-chapter-lead">把可外推范围、护栏和政策动作连接到具体证据。</p></header><div class="full-chapter-body"><p><strong>目标人群:</strong>企业客服人员,按资历和基线技能分层。</p>
|
|
2151
2151
|
<p><strong>目标情境:</strong>使用经审核知识库、由人工监督的客服流程。</p>
|
|
2152
2152
|
<p><strong>适用条件:</strong></p><ul><li>从低风险且知识库覆盖的队列开始;客服审查并编辑每条建议回复。</li><li>按资历和基线技能分层;保留专家判断并记录复核工作量。</li><li>试点前落实客户数据最小化与脱敏、访问和留存限制,并审查供应商数据使用条款。</li><li>陌生、模糊、敏感或高风险请求交由合格人员处理;禁止自动作出承诺。</li></ul><div class="boundary-block"><h3>不可外推的结论</h3><ul><li>声称普遍收益或自动部署安全超出证据边界:没有纳入研究测量隐私事件、本地净成本或亚组服务质量。</li></ul></div><div class="evidence-to-action" data-visual="evidence-to-action"><article class="action-node action-evidence"><span>证据</span><p class="action-node-text">在有边界的客服场景中,AI 辅助可缩短会话处理时间;吞吐量是另一项指标。 — E-001</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-applicability"><span>适用性</span><p class="action-node-text">从低风险且知识库覆盖的队列开始;客服审查并编辑每条建议回复。; 按资历和基线技能分层;保留专家判断并记录复核工作量。; 试点前落实客户数据最小化与脱敏、访问和留存限制,并审查供应商数据使用条款。; 陌生、模糊、敏感或高风险请求交由合格人员处理;禁止自动作出承诺。</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-decision"><span>决策</span><p class="action-node-text">试点验证</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-guardrails"><span>护栏</span><p class="action-node-text">从低风险且知识库覆盖的队列开始;客服审查并编辑每条建议回复。</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-stop"><span>停止条件</span><p class="action-node-text">一旦核实隐私泄漏、严重不安全建议或越权客户承诺,立即暂停,调查后再决定是否恢复。; 若盲评质量越过预先约定的非劣界值(包括各资历分层),暂停扩展。</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-evaluation"><span>评价</span><p class="action-node-text">仅在质量非劣、每付薪工时有效解决量提高、净成本可接受且无未解决严重安全事件时考虑扩展;阈值须在本地预先约定。</p></article></div><p><strong>目标人群:</strong>企业客服人员,按资历和基线技能分层。 · <strong>试点时长:</strong>建议:两周基线和六周试点;尚未执行。</p>
|
|
2153
|
-
<p><strong
|
|
2153
|
+
<p><strong>干预使用规则:</strong>从低风险且知识库覆盖的队列开始;客服审查并编辑每条建议回复。</p>
|
|
2154
2154
|
<h3>停止条件</h3><ul><li>一旦核实隐私泄漏、严重不安全建议或越权客户承诺,立即暂停,调查后再决定是否恢复。</li><li>若盲评质量越过预先约定的非劣界值(包括各资历分层),暂停扩展。</li></ul>
|
|
2155
|
-
<h3
|
|
2156
|
-
<svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="
|
|
2155
|
+
<h3>干预方案时间线信息图</h3>
|
|
2156
|
+
<svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="干预方案时间线"><title>干预方案时间线</title><desc>各试点阶段的短名称与活动数量;完整干预使用规则见阶段说明块。</desc><defs><marker id="arr-zh-full-intervention" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">干预方案时间线</text><rect x="24" y="100" width="160" height="96" rx="8" fill="#B8694A"/><text x="104.0" y="148.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="104.0" y="148.0">Phase 1</tspan></text></svg></div></section><section id="full-05-evaluation" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>05 试点设计、评估与停止条件</h2><p class="full-chapter-lead">用独立效果测量验证试点,并预先写清停止条件。</p></header><div class="full-chapter-body"><p><strong>研究问题:</strong>企业客服团队是否应引入生成式 AI 助手?</p>
|
|
2157
2157
|
<p><strong>基线:</strong>分配前记录解决率、付薪工时、重复联系、盲评质量及复核成本。</p>
|
|
2158
|
-
<p><strong
|
|
2158
|
+
<p><strong>试点结束:</strong>试点结束时评估相同结果;独立审计隐私与不安全承诺。</p>
|
|
2159
2159
|
<p><strong>成功阈值:</strong>仅在质量非劣、每付薪工时有效解决量提高、净成本可接受且无未解决严重安全事件时考虑扩展;阈值须在本地预先约定。</p>
|
|
2160
2160
|
<p><strong>分析计划:</strong>入组前预注册意向治疗比较、团队聚类不确定性及亚组估计。根据本地基线变异和业务目标确定样本量与非劣界值。</p>
|
|
2161
2161
|
<h3>评价设计信息图</h3>
|
|
2162
|
-
<svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="评价设计流程"><title>评价设计流程</title><desc
|
|
2162
|
+
<svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="评价设计流程"><title>评价设计流程</title><desc>基线、试点结束、持续监测与扩展评估的评价流程;完整指标与分析计划见评估章节。</desc><defs><marker id="arr-zh-full-evaluation" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">评价设计流程</text><rect x="30" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="105.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="105.0" y="138.0">基线</tspan></text><text x="105.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">前测</text><line x1="180" y1="138" x2="192" y2="138" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-evaluation)"/><rect x="192" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="267.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="267.0" y="138.0">后测</tspan></text><text x="267.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">后测</text><line x1="342" y1="138" x2="354" y2="138" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-evaluation)"/><rect x="354" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="429.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="429.0" y="138.0">保持</tspan></text><text x="429.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">持续监测</text><line x1="504" y1="138" x2="516" y2="138" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-zh-full-evaluation)"/><rect x="516" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="591.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="591.0" y="138.0">迁移</tspan></text><text x="591.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">扩展评估</text></svg><div class="visual-suppressed"><strong>基准图已抑制</strong><p>result.json 未携带 benchmark.baselines,本图不绘制;基准表现见独立基准报告。</p></div></div></section><section id="full-06-sources" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>06 来源、溯源与附录</h2><p class="full-chapter-lead">保留原始来源、URL、证据 ID 和获取信息,确保可回查。</p></header><div class="full-chapter-body"><h3>来源列表</h3><div class='table-wrap'><table class='data-table source-table'><thead><tr><th>ID</th><th>标题</th><th>年份</th><th>权威级别</th><th>可验证位置</th></tr></thead><tbody><tr><td><code>S-001</code></td><td class='cell-main source-title-cell'>Generative AI at Work<details class="source-expander detail-expander"><summary>查看来源与溯源</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>原文标题</dt><dd>Generative AI at Work</dd></div><div class="source-detail-row"><dt>年份</dt><dd>2025</dd></div><div class="source-detail-row"><dt>权威级别</dt><dd>T1 DOI 可验证论文</dd></div><div class="source-detail-row"><dt>获取方式</dt><dd>builtin</dd></div><div class="source-detail-row"><dt>获取状态</dt><dd>FETCH_VALID</dd></div><div class="source-detail-row"><dt>可验证链接</dt><dd><a href="https://academic.oup.com/qje/article/140/2/889/7990658">https://academic.oup.com/qje/article/140/2/889/7990658</a></dd></div></dl></details></td><td>2025</td><td>T1 DOI 可验证论文</td><td class='cell-main'><a href='https://academic.oup.com/qje/article/140/2/889/7990658'>https://academic.oup.com/qje/article/140/2/889/7990658</a></td></tr><tr><td><code>S-002</code></td><td class='cell-main source-title-cell'>Experimental evidence on the productivity effects of generative artificial intelligence<details class="source-expander detail-expander"><summary>查看来源与溯源</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>原文标题</dt><dd>Experimental evidence on the productivity effects of generative artificial intelligence</dd></div><div class="source-detail-row"><dt>年份</dt><dd>2023</dd></div><div class="source-detail-row"><dt>权威级别</dt><dd>T1 DOI 可验证论文</dd></div><div class="source-detail-row"><dt>获取方式</dt><dd>builtin</dd></div><div class="source-detail-row"><dt>获取状态</dt><dd>FETCH_VALID</dd></div><div class="source-detail-row"><dt>可验证链接</dt><dd><a href="https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf">https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf</a></dd></div></dl></details></td><td>2023</td><td>T1 DOI 可验证论文</td><td class='cell-main'><a href='https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf'>https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf</a></td></tr><tr><td><code>S-003</code></td><td class='cell-main source-title-cell'>Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality<details class="source-expander detail-expander"><summary>查看来源与溯源</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>原文标题</dt><dd>Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality</dd></div><div class="source-detail-row"><dt>年份</dt><dd>2026</dd></div><div class="source-detail-row"><dt>权威级别</dt><dd>T1 DOI 可验证论文</dd></div><div class="source-detail-row"><dt>获取方式</dt><dd>builtin</dd></div><div class="source-detail-row"><dt>获取状态</dt><dd>FETCH_VALID</dd></div><div class="source-detail-row"><dt>可验证链接</dt><dd><a href="https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838">https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838</a></dd></div></dl></details></td><td>2026</td><td>T1 DOI 可验证论文</td><td class='cell-main'><a href='https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838'>https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838</a></td></tr></tbody></table></div><h3>Fetch 溯源</h3><div class='table-wrap'><table class='data-table'><thead><tr><th>来源</th><th>Fetch 方式</th><th>状态</th><th>降级</th><th>时间</th></tr></thead><tbody><tr><td><code>S-001</code></td><td>builtin</td><td>FETCH_VALID</td><td></td><td></td></tr><tr><td><code>S-002</code></td><td>builtin</td><td>FETCH_VALID</td><td></td><td></td></tr><tr><td><code>S-003</code></td><td>builtin</td><td>FETCH_VALID</td><td></td><td></td></tr></tbody></table></div></div></section></main></div>
|
|
2163
2163
|
</div>
|
|
2164
2164
|
<footer class="report-footer"><p>EduEvidence 证据报告 · Schema PASS · Claim Binding PASS · Numeric Consistency PASS · Bilingual Structure PASS · 语言人话化 PASS · 无伪精度 PASS · Lieflat 数据溯源 PASS · 坐标轴无失真 NOT_CHECKED · 色盲安全 NOT_CHECKED · 单文件离线可打开 · 数据源:result.json</p></footer>
|
|
2165
2165
|
</div>
|
|
@@ -2173,7 +2173,7 @@ select:focus-visible,
|
|
|
2173
2173
|
<button type="button" class="report-view-btn" data-report-view="full" aria-pressed="false">Full Report</button>
|
|
2174
2174
|
</nav>
|
|
2175
2175
|
</header>
|
|
2176
|
-
<div class="report-page report-page-brief" data-report-page="brief"><nav class="brief-navigation" aria-label="Brief navigation"><a href="#brief-decision-en"><span>01</span>Decision first</a><a href="#brief-outcomes-en"><span>02</span>
|
|
2176
|
+
<div class="report-page report-page-brief" data-report-page="brief"><nav class="brief-navigation" aria-label="Brief navigation"><a href="#brief-decision-en"><span>01</span>Decision first</a><a href="#brief-outcomes-en"><span>02</span>Process output ≠ policy effect</a><a href="#brief-tribunal-en"><span>03</span>Evidence tribunal</a><a href="#brief-action-en"><span>04</span>Evidence to action</a><a href="#brief-sources-en"><span>05</span>Key sources</a></nav><div class="brief-reading-content"><section class="brief-block brief-decision" id="brief-decision-en"><header class="brief-block-header"><h2>Decision first</h2><p>What to do, how confident we are, and the most important evidence boundary.</p></header><div class="brief-block-body">
|
|
2177
2177
|
<div class="decision-hero pilot" data-visual="decision-hero">
|
|
2178
2178
|
<div class="hero-decision">
|
|
2179
2179
|
<span class="eyebrow">Recommended decision</span>
|
|
@@ -2188,7 +2188,7 @@ select:focus-visible,
|
|
|
2188
2188
|
<article class="hero-insight next"><span>Next action</span><p class="hero-insight-text">Run a supervised pilot on approved knowledge bases with human review on every reply, and track escalation and correction rates.</p></article>
|
|
2189
2189
|
</div>
|
|
2190
2190
|
<p class="hero-provenance"><span>Evidence / sources</span> · 4 / 3</p>
|
|
2191
|
-
</div></div></section><section class="brief-block brief-outcomes" id="brief-outcomes-en"><header class="brief-block-header"><h2>
|
|
2191
|
+
</div></div></section><section class="brief-block brief-outcomes" id="brief-outcomes-en"><header class="brief-block-header"><h2>Process output ≠ policy effect</h2><p>Only informative outcome separation; positive, negative and null effects use effect_direction.</p></header><div class="brief-block-body"><div class="outcome-separation" data-visual="outcome-separation"><div class="visual-heading"><h3>Outcome Separation · Process output ≠ policy effect</h3><p>Outcomes are adjudicated separately so pilot throughput or efficiency gains are not silently treated as real policy effects.</p><p class="semantic-note">Effect direction comes from evidence.effect_direction; supporting a claim does not imply a positive outcome.</p></div><div class="outcome-groups"><article class="outcome-group outcome-learning"><h3>Effect / target outcome</h3><ul><li><strong>Policy effectiveness</strong><span class="outcome-states"><span class="dir pos">Positive effect 2</span></span></li></ul></article><article class="outcome-group outcome-risk"><h3>Risk / implementation risk</h3><ul><li><strong>Implementation risk</strong><span class="outcome-states"><span class="dir neg">Negative effect 2</span></span></li></ul></article></div></div></div></section><section class="brief-block brief-tribunal" id="brief-tribunal-en"><header class="brief-block-header"><h2>Evidence tribunal</h2><p>Supported, uncertain, contradicted and missing evidence stay separated instead of flattened into long prose.</p></header><div class="brief-block-body"><div class="evidence-tribunal" data-visual="evidence-tribunal-grid"><p class="tribunal-summary"><strong>Decision: </strong>Pilot · <strong>Confidence: </strong>Moderate</p><div class="tribunal-grid"><article class="tribunal-card supported"><header><span aria-hidden="true">✓</span><h3>Can claim</h3><span class="tribunal-count">4</span></header><ul><li><p>AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome. — E-001</p><div class="tribunal-evidence-refs"><code>E-001</code></div></li><li><p>ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support. — E-002</p><div class="tribunal-evidence-refs"><code>E-002</code></div></li><li><p>AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support. — E-003</p><div class="tribunal-evidence-refs"><code>E-003</code></div></li></ul><details class="tribunal-more"><summary>View 1 more</summary><ul><li><p>Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits. — E-004</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li></ul></details></article><article class="tribunal-card uncertain"><header><span aria-hidden="true">?</span><h3>Cannot yet claim</h3><span class="tribunal-count">1</span></header><ul><li><p>Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set [no direct evidence].</p></li></ul></article><article class="tribunal-card contradicted"><header><span aria-hidden="true">×</span><h3>Contradicted claims</h3><span class="tribunal-count">0</span></header><ul><li>No data.</li></ul></article><article class="tribunal-card missing"><header><span aria-hidden="true">…</span><h3>Missing evidence</h3><span class="tribunal-count">1</span></header><ul><li><p>Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set [no direct evidence].</p></li></ul></article></div></div></div></section><section class="brief-block brief-action" id="brief-action-en"><header class="brief-block-header"><h2>Evidence to action</h2><p>Applicability, guardrails, stop conditions and evaluation form one executable path.</p></header><div class="brief-block-body"><div class="evidence-to-action" data-visual="evidence-to-action"><article class="action-node action-evidence"><span>Evidence</span><p class="action-node-text">AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome. — E-001</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-applicability"><span>Applicability</span><div class="action-node-text"><p>Start with low-risk, knowledge-base-covered queues; agents inspect and edit every proposed reply.; Stratify by tenure and baseline skill;…</p><details class="detail-expander"><summary>Expand full explanation</summary><div class="detail-body"><p>Start with low-risk, knowledge-base-covered queues; agents inspect and edit every proposed reply.; Stratify by tenure and baseline skill; retain expert discretion and measure review workload.; Minimize and redact customer data, restrict access and retention, and review supplier data-use terms before a pilot.; Escalate unfamiliar, ambiguous, sensitive or high-stakes requests to qualified humans; no autonomous commitments.</p></div></details></div></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-decision"><span>Decision</span><p class="action-node-text">Pilot</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-guardrails"><span>Guardrails</span><p class="action-node-text">Start with low-risk, knowledge-base-covered queues; agents inspect and edit every proposed reply.</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-stop"><span>Stop conditions</span><div class="action-node-text"><p>Pause immediately after a verified privacy leak or a serious unsafe or unauthorized customer commitment; investigate before resuming.;…</p><details class="detail-expander"><summary>Expand full explanation</summary><div class="detail-body"><p>Pause immediately after a verified privacy leak or a serious unsafe or unauthorized customer commitment; investigate before resuming.; Pause expansion if blinded quality review crosses a pre-agreed noninferiority margin, including within experience strata.</p></div></details></div></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-evaluation"><span>Evaluation</span><div class="action-node-text"><p>Expand only if quality is noninferior, verified resolutions per paid hour improve, net cost is acceptable and no serious unresolved safety incident re…</p><details class="detail-expander"><summary>Expand full explanation</summary><div class="detail-body"><p>Expand only if quality is noninferior, verified resolutions per paid hour improve, net cost is acceptable and no serious unresolved safety incident remains; thresholds require local agreement.</p></div></details></div></article></div></div></section><section class="brief-block brief-lieflat" id="brief-lieflat-en"><header class="brief-block-header"><h2>Lieflat Editorial Gallery</h2><p>Charts selected and composed by AI from the Lieflat catalog; every number traces back to result.json.</p></header><div class="brief-block-body"><div class="lieflat-gallery-container"><figure class="lieflat-card" data-lieflat data-visual="lieflat-bubble_almanac" data-chart-id="lieflat-bubble-almanac.svg"><h3 class="lieflat-title">Year × dimension evidence almanac</h3><p class="lieflat-sub">Bubble area ∝ study count · solid core = significant results</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 540 300" width="100%" height="100%" role="img" aria-label="Year × dimension evidence almanac" style="background:#FAFAFA;">
|
|
2192
2192
|
<line x1="44" y1="70.0" x2="520" y2="70.0" stroke="#E2E8F0" stroke-width="0.5" class="lf-fade" style="--motion-delay:0ms"/>
|
|
2193
2193
|
<line x1="44" y1="77.0" x2="520" y2="77.0" stroke="#E2E8F0" stroke-width="0.5" class="lf-fade" style="--motion-delay:14ms"/>
|
|
2194
2194
|
<line x1="44" y1="84.0" x2="520" y2="84.0" stroke="#E2E8F0" stroke-width="0.5" class="lf-fade" style="--motion-delay:28ms"/>
|
|
@@ -2217,27 +2217,27 @@ select:focus-visible,
|
|
|
2217
2217
|
<line x1="44" y1="245.0" x2="520" y2="245.0" stroke="#E2E8F0" stroke-width="0.5" class="lf-fade" style="--motion-delay:350ms"/>
|
|
2218
2218
|
<line x1="44" y1="252.0" x2="520" y2="252.0" stroke="#E2E8F0" stroke-width="0.5" class="lf-fade" style="--motion-delay:364ms"/>
|
|
2219
2219
|
<line x1="44" y1="259.0" x2="520" y2="259.0" stroke="#E2E8F0" stroke-width="0.5" class="lf-fade" style="--motion-delay:378ms"/>
|
|
2220
|
-
<text x="150" y="76" fill="#0F172A" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:40ms">
|
|
2221
|
-
<text x="484.8" y="76" fill="#0F172A" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:140ms">
|
|
2220
|
+
<text x="150" y="76" fill="#0F172A" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:40ms">Policy effecti</text>
|
|
2221
|
+
<text x="484.8" y="76" fill="#0F172A" font-size="9" font-weight="600" text-anchor="middle" class="lf-fade" style="--motion-delay:140ms">Implementation</text>
|
|
2222
2222
|
<text x="96" y="96.0" fill="#475569" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:80ms">2023</text>
|
|
2223
2223
|
<text x="96" y="140.0" fill="#475569" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:180ms">2025</text>
|
|
2224
2224
|
<text x="96" y="184.0" fill="#475569" font-size="9" font-weight="800" text-anchor="end" class="lf-fade" style="--motion-delay:280ms">2026</text>
|
|
2225
|
-
<circle cx="150.0" cy="92.0" r="3.6" fill="#0F172A" fill-opacity="0.22" stroke="#0F172A" stroke-width="1.2" class="lf-pop" style="--motion-delay:100ms"><title>
|
|
2226
|
-
<circle cx="150.0" cy="136.0" r="3.6" fill="#0F172A" fill-opacity="0.22" stroke="#0F172A" stroke-width="1.2" class="lf-pop" style="--motion-delay:112ms"><title>
|
|
2227
|
-
<circle cx="520.0" cy="136.0" r="3.6" fill="#0F172A" fill-opacity="0.22" stroke="#0F172A" stroke-width="1.2" class="lf-pop" style="--motion-delay:124ms"><title>
|
|
2228
|
-
<circle cx="520.0" cy="180.0" r="3.6" fill="#0F172A" fill-opacity="0.22" stroke="#0F172A" stroke-width="1.2" class="lf-pop" style="--motion-delay:136ms"><title>
|
|
2225
|
+
<circle cx="150.0" cy="92.0" r="3.6" fill="#0F172A" fill-opacity="0.22" stroke="#0F172A" stroke-width="1.2" class="lf-pop" style="--motion-delay:100ms"><title>Policy effectiveness (2023) — N = 1 studies, significant = 0</title></circle>
|
|
2226
|
+
<circle cx="150.0" cy="136.0" r="3.6" fill="#0F172A" fill-opacity="0.22" stroke="#0F172A" stroke-width="1.2" class="lf-pop" style="--motion-delay:112ms"><title>Policy effectiveness (2025) — N = 1 studies, significant = 0</title></circle>
|
|
2227
|
+
<circle cx="520.0" cy="136.0" r="3.6" fill="#0F172A" fill-opacity="0.22" stroke="#0F172A" stroke-width="1.2" class="lf-pop" style="--motion-delay:124ms"><title>Implementation risk (2025) — N = 1 studies, significant = 0</title></circle>
|
|
2228
|
+
<circle cx="520.0" cy="180.0" r="3.6" fill="#0F172A" fill-opacity="0.22" stroke="#0F172A" stroke-width="1.2" class="lf-pop" style="--motion-delay:136ms"><title>Implementation risk (2026) — N = 1 studies, significant = 0</title></circle>
|
|
2229
2229
|
</svg></div><figcaption class="lieflat-caption">Drawn only when years and outcome dimensions exist.</figcaption><p class="lieflat-src">L9 Bubble Almanac · Evidence.year X Dimension</p></figure><figure class="lieflat-card" data-lieflat data-visual="lieflat-matrix_heat" data-chart-id="lieflat-matrix-heat.svg"><h3 class="lieflat-title">Year × outcome evidence density</h3><p class="lieflat-sub">Each cell counts evidence items for that year and outcome</p><div class="lieflat-figure"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 540 200" width="100%" height="100%" role="img" aria-label="Year × outcome evidence density" style="background:#FAFAFA;">
|
|
2230
2230
|
<text x="211.7" y="66" text-anchor="middle" font-size="9" font-weight="800" fill="#475569" class="lf-fade" style="--motion-delay:40ms">2023</text>
|
|
2231
2231
|
<text x="335.0" y="66" text-anchor="middle" font-size="9" font-weight="800" fill="#475569" class="lf-fade" style="--motion-delay:140ms">2025</text>
|
|
2232
2232
|
<text x="458.3" y="66" text-anchor="middle" font-size="9" font-weight="800" fill="#475569" class="lf-fade" style="--motion-delay:240ms">2026</text>
|
|
2233
|
-
<text x="138" y="97.0" text-anchor="end" font-size="8" font-weight="600" fill="#475569" class="lf-fade" style="--motion-delay:60ms">
|
|
2233
|
+
<text x="138" y="97.0" text-anchor="end" font-size="8" font-weight="600" fill="#475569" class="lf-fade" style="--motion-delay:60ms">Policy effecti</text>
|
|
2234
2234
|
<rect x="152.0" y="80.0" width="119.3" height="28" rx="4" fill="#0F172A" fill-opacity="0.90" class="lf-pop" style="--motion-delay:0ms"/>
|
|
2235
2235
|
<text x="211.7" y="97.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#0F172A" class="lf-pop" style="--motion-delay:60ms">1</text>
|
|
2236
2236
|
<rect x="275.3" y="80.0" width="119.3" height="28" rx="4" fill="#0F172A" fill-opacity="0.90" class="lf-pop" style="--motion-delay:12ms"/>
|
|
2237
2237
|
<text x="335.0" y="97.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#0F172A" class="lf-pop" style="--motion-delay:72ms">1</text>
|
|
2238
2238
|
<rect x="398.7" y="80.0" width="119.3" height="28" rx="4" fill="#E2E8F0" fill-opacity="0.5" class="lf-fade" style="--motion-delay:24ms"/>
|
|
2239
2239
|
<text x="458.3" y="97.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#0F172A" class="lf-pop" style="--motion-delay:84ms">0</text>
|
|
2240
|
-
<text x="138" y="129.0" text-anchor="end" font-size="8" font-weight="600" fill="#475569" class="lf-fade" style="--motion-delay:160ms">
|
|
2240
|
+
<text x="138" y="129.0" text-anchor="end" font-size="8" font-weight="600" fill="#475569" class="lf-fade" style="--motion-delay:160ms">Implementation</text>
|
|
2241
2241
|
<rect x="152.0" y="112.0" width="119.3" height="28" rx="4" fill="#E2E8F0" fill-opacity="0.5" class="lf-fade" style="--motion-delay:12ms"/>
|
|
2242
2242
|
<text x="211.7" y="129.0" text-anchor="middle" font-size="8.0" font-weight="800" fill="#0F172A" class="lf-pop" style="--motion-delay:72ms">0</text>
|
|
2243
2243
|
<rect x="275.3" y="112.0" width="119.3" height="28" rx="4" fill="#0F172A" fill-opacity="0.90" class="lf-pop" style="--motion-delay:24ms"/>
|
|
@@ -2394,7 +2394,7 @@ select:focus-visible,
|
|
|
2394
2394
|
</svg></div><figcaption class="lieflat-caption">Shows which checklist items concentrate the weaknesses.</figcaption><p class="lieflat-src">L15 Ballot Tally · Methodology.flag Rates</p></figure></div></div></section><section class="brief-block brief-sources" id="brief-sources-en"><header class="brief-block-header"><h2>Key sources</h2><p>Only the key sources in the brief; full traceability expands in the full report.</p></header><div class="brief-block-body"><div class="brief-source-grid"><article class="brief-source"><code>S-001</code><h3><a href="https://academic.oup.com/qje/article/140/2/889/7990658">Generative AI at Work</a></h3><p>Tier 1 DOI-verified paper · 2025</p></article><article class="brief-source"><code>S-002</code><h3><a href="https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf">Experimental evidence on the productivity effects of generative artificial intelligence</a></h3><p>Tier 1 DOI-verified paper · 2023</p></article><article class="brief-source"><code>S-003</code><h3><a href="https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838">Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality</a></h3><p>Tier 1 DOI-verified paper · 2026</p></article></div></div></section></div></div>
|
|
2395
2395
|
<div class="report-page report-page-full" data-report-page="full" hidden>
|
|
2396
2396
|
<div class="full-report-intro"><h2>Full Report</h2><p>Conclusions first: every traceable piece of evidence and method note lives here, with visuals only at points where they add meaning. Every number traces back to result.json.</p></div>
|
|
2397
|
-
<div class="full-report-layout"><aside class="full-report-toc" aria-label="Contents"><div class="toc-head"><strong>Contents</strong><button type="button" class="toc-collapse" aria-expanded="true" data-label-collapse="Collapse contents" data-label-expand="Expand contents">Collapse contents</button></div><nav><a href="#full-01-decision-en" data-toc-target="full-01-decision-en" data-chapter-key="decision">01 Decision, Adjudication & Research Boundary</a><a href="#full-02-evidence-en" data-toc-target="full-02-evidence-en" data-chapter-key="evidence">02 Key Evidence & Outcome Separation</a><a href="#full-03-quality-en" data-toc-target="full-03-quality-en" data-chapter-key="quality">03 Evidence Quality, Counterevidence & Method Audit</a><a href="#full-04-action-en" data-toc-target="full-04-action-en" data-chapter-key="action">04 Applicability &
|
|
2397
|
+
<div class="full-report-layout"><aside class="full-report-toc" aria-label="Contents"><div class="toc-head"><strong>Contents</strong><button type="button" class="toc-collapse" aria-expanded="true" data-label-collapse="Collapse contents" data-label-expand="Expand contents">Collapse contents</button></div><nav><a href="#full-01-decision-en" data-toc-target="full-01-decision-en" data-chapter-key="decision">01 Decision, Adjudication & Research Boundary</a><a href="#full-02-evidence-en" data-toc-target="full-02-evidence-en" data-chapter-key="evidence">02 Key Evidence & Outcome Separation</a><a href="#full-03-quality-en" data-toc-target="full-03-quality-en" data-chapter-key="quality">03 Evidence Quality, Counterevidence & Method Audit</a><a href="#full-04-action-en" data-toc-target="full-04-action-en" data-chapter-key="action">04 Applicability & Policy Action</a><a href="#full-05-evaluation-en" data-toc-target="full-05-evaluation-en" data-chapter-key="evaluation">05 Pilot, Evaluation & Stop Conditions</a><a href="#full-06-sources-en" data-toc-target="full-06-sources-en" data-chapter-key="sources">06 Sources, Traceability & Appendix</a></nav></aside><main class="full-report-content"><section id="full-01-decision-en" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>01 Decision, Adjudication & Research Boundary</h2><p class="full-chapter-lead">State the final adjudication and research boundary before explaining why.</p></header><div class="full-chapter-body">
|
|
2398
2398
|
<div class="decision-hero pilot" data-visual="decision-hero">
|
|
2399
2399
|
<div class="hero-decision">
|
|
2400
2400
|
<span class="eyebrow">Recommended decision</span>
|
|
@@ -2409,19 +2409,19 @@ select:focus-visible,
|
|
|
2409
2409
|
<article class="hero-insight next"><span>Next action</span><p class="hero-insight-text">Run a supervised pilot on approved knowledge bases with human review on every reply, and track escalation and correction rates.</p></article>
|
|
2410
2410
|
</div>
|
|
2411
2411
|
<p class="hero-provenance"><span>Evidence / sources</span> · 4 / 3</p>
|
|
2412
|
-
</div><div class="scope-grid"><article class="scope-card"><h3>Research question</h3><p class="scope-text">Should an enterprise customer-support team introduce a generative AI assistant?</p></article><article class="scope-card"><h3>AI intervention</h3><p class="scope-text">Policy: Human Supervised Customer Support Assistant; Policy type: Institutional reform; Mechanism: Approved knowledge base + agent review</p></article><article class="scope-card"><h3>Comparison</h3><p class="scope-text">Existing support workflow without generative AI suggestions.</p></article><article class="scope-card"><h3>Outcome constructs</h3><p class="scope-text">Primary outcomes: Policy Effectiveness, Implementation Risk; Secondary outcomes: Cost Effectiveness, equity, feasibility</p></article><article class="scope-card"><h3>Research scope</h3><p class="scope-text">Time range: 2023–2026; Evidence Types: Quasi-experimental, Randomised controlled trial</p></article><article class="scope-card"><h3>Decision success condition</h3><p class="scope-text">Improve verified resolutions per paid staff hour while preserving service quality, privacy and staff autonomy.</p></article></div></div></section><section id="full-02-evidence-en" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>02 Key Evidence & Outcome Separation</h2><p class="full-chapter-lead">Place task performance, actual learning, retention and risk on one evidence map without conflating them.</p></header><div class="full-chapter-body"><div class="retrieval-grid"><article><h3>Inclusion criteria</h3><ul></ul></article><article><h3>Exclusion criteria</h3><ul></ul></article></div><div class="retrieval-coverage"><h3>Source coverage</h3><p><code>S-001</code> <code>S-002</code> <code>S-003</code></p><p class="retrieval-note">This report shows only retrieval metadata present in result; it does not fabricate PRISMA/funnel counts when none exist.</p></div><div class="outcome-separation" data-visual="outcome-separation"><div class="visual-heading"><h3>Outcome Separation · Task performance ≠ learning</h3><p>Outcomes are adjudicated separately so faster training performance is not silently treated as evidence of learning.</p><p class="semantic-note">Effect direction comes from evidence.effect_direction; supporting a claim does not imply a positive outcome.</p></div><div class="outcome-groups"><article class="outcome-group outcome-other"><h3>Other outcomes</h3><ul><li><strong>Policy Effectiveness</strong><span class="outcome-states"><span class="dir pos">Positive effect 2</span></span></li><li><strong>Implementation Risk</strong><span class="outcome-states"><span class="dir neg">Negative effect 2</span></span></li></ul></article></div></div><div class='table-wrap outcome-table'><table class='data-table'><thead><tr><th>Outcome</th><th>Positive effect</th><th>Negative effect</th><th>Null effect</th><th>Evidence</th></tr></thead><tbody><tr><td><strong>Policy Effectiveness</strong><span class='raw-tag' title='raw id'>policy_effectiveness</span></td><td class='num'>2</td><td class='num'>0</td><td class='num'>0</td><td><code>E-001</code> <code>E-002</code> </td></tr><tr><td><strong>Implementation Risk</strong><span class='raw-tag' title='raw id'>implementation_risk</span></td><td class='num'>0</td><td class='num'>2</td><td class='num'>0</td><td><code>E-003</code> <code>E-004</code> </td></tr></tbody></table></div><div class='matrix-controls'><div class='matrix-tools'><input id='matrix-search-full-en' type='search' placeholder='Search evidence…' aria-label='Filter / search evidence'><select id='matrix-direction-full-en' aria-label='Filter by effect direction'><option value=''>All effects</option><option value='positive'>Positive effect</option><option value='negative'>Negative effect</option><option value='null'>Null effect</option></select><select id='matrix-outcome-full-en' aria-label='Filter by outcome type'><option value=''>All outcomes</option><option value='implementation_risk'>Implementation Risk</option><option value='policy_effectiveness'>Policy Effectiveness</option></select></div></div><div class='table-wrap matrix-wrap'><table id='evidence-matrix-full-en' class='data-table evidence-matrix'><thead><tr><th>ID</th><th>Outcome</th><th>Effect</th><th>Quality</th><th>Claim</th><th>Source</th></tr></thead><tbody><tr data-effect="positive" data-direction="positive" data-outcome="policy_effectiveness" data-search="e-001 st-001 sample-support-all generative ai at work ai assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome. customer-support agents positive s-001"><td><code>E-001</code></td><td><strong>Policy Effectiveness</strong></td><td><span class="dir pos">Positive effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">9</span><span class="quality-meter" aria-hidden="true"><i style="width:90%"></i></span></div></td><td class="claim-cell"><p>AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>ST-001</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SAMPLE-support-all</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Generative AI at Work</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Generative AI at Work</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Quasi-experimental</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>Customer-support agents</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>5172</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>Chat handling time; separate throughput summary: about 15% more issues resolved per hour.</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Positive effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>One firm, one tool and a staggered nonrandom rollout; causal interpretation depends on identification assumptions.</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 1 · D2 Sample quality = 2 · D3 Measurement validity = 2 · D4 Temporal strength = 2 · D5 Directness = 2</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>9.0</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://academic.oup.com/qje/article/140/2/889/7990658</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://academic.oup.com/qje/article/140/2/889/7990658">https://academic.oup.com/qje/article/140/2/889/7990658</a></dd></div></dl></details></td><td><a class="source-link" href="https://academic.oup.com/qje/article/140/2/889/7990658" title="Generative AI at Work"><code>S-001</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="policy_effectiveness" data-search="e-002 st-002 sample-writing experimental evidence on the productivity effects of generative artificial intelligence chatgpt reduced time on short professional writing tasks; this is indirect evidence for customer support. college-educated working professionals positive s-002"><td><code>E-002</code></td><td><strong>Policy Effectiveness</strong></td><td><span class="dir pos">Positive effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>ST-002</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SAMPLE-writing</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Experimental evidence on the productivity effects of generative artificial intelligence</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Experimental evidence on the productivity effects of generative artificial intelligence</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Randomized controlled trial</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>College-educated working professionals</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>453</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>Self-reported task time: approximately 40% lower; assessed writing quality approximately 18% higher is a separate measure.</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Positive effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>Indirect evidence, not directly generalizable: brief incentivized writing tasks did not demand precise factual accuracy or customer-specific context.</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 2 · D2 Sample quality = 2 · D3 Measurement validity = 1 · D4 Temporal strength = 1 · D5 Directness = 1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf">https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf</a></dd></div></dl></details></td><td><a class="source-link" href="https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf" title="Experimental evidence on the productivity effects of generative artificial intelligence"><code>S-002</code></a></td></tr><tr data-effect="negative" data-direction="negative" data-outcome="implementation_risk" data-search="e-003 st-003 sample-consulting-outside navigating the jagged technological frontier: field experimental evidence of the effects of artificial intelligence on knowledge worker productivity and quality ai can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support. bcg consultants in the outside-frontier experiment negative s-003"><td><code>E-003</code></td><td><strong>Implementation Risk</strong></td><td><span class="dir neg">Negative effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>ST-003</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SAMPLE-consulting-outside</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2026</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Randomized controlled trial</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>BCG consultants in the outside-frontier experiment</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>373</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>Correct business recommendation: about 19 percentage points lower across AI arms; Table 7 has 373 participants, not 758.</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Negative effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>Indirect evidence, not directly generalizable: consultants solving an experimental business case, not live customer tickets.</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 2 · D2 Sample quality = 2 · D3 Measurement validity = 2 · D4 Temporal strength = 1 · D5 Directness = 1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838">https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838</a></dd></div></dl></details></td><td><a class="source-link" href="https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838" title="Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality"><code>S-003</code></a></td></tr><tr data-effect="negative" data-direction="negative" data-outcome="implementation_risk" data-search="e-004 st-001 sample-support-all generative ai at work experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits. experienced and high-skill customer-support agents negative s-001"><td><code>E-004</code></td><td><strong>Implementation Risk</strong></td><td><span class="dir neg">Negative effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>ST-001</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SAMPLE-support-all</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Generative AI at Work</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Generative AI at Work</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Quasi-experimental</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>Experienced and high-skill customer-support agents</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>Small quality declines among the most experienced and highest-skilled support staff.</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Negative effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>Same study as E-001; subgroup sample size was not extracted. This is not an independent replication.</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 1 · D2 Sample quality = 1 · D3 Measurement validity = 2 · D4 Temporal strength = 2 · D5 Directness = 2</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://academic.oup.com/qje/article/140/2/889/7990658</dd></div><div class="evidence-detail-row"><dt>Verifiable link</dt><dd><a href="https://academic.oup.com/qje/article/140/2/889/7990658">https://academic.oup.com/qje/article/140/2/889/7990658</a></dd></div></dl></details></td><td><a class="source-link" href="https://academic.oup.com/qje/article/140/2/889/7990658" title="Generative AI at Work"><code>S-001</code></a></td></tr></tbody></table></div></div></section><section id="full-03-quality-en" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>03 Evidence Quality, Counterevidence & Method Audit</h2><p class="full-chapter-lead">Examine why evidence is credible, where it conflicts, and which conclusions require downgrading.</p></header><div class="full-chapter-body"><section class="method-review-group"><header class="method-review-title"><h3>Audit target: overall</h3><span class="method-verdict">Concern</span></header><div class="method-audit-grid"><article class="method-audit-item method-met"><div class="method-audit-head"><strong>Evidence Level</strong><span class="method-status">Met</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Causal Identification</strong><span class="method-status">Partial</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>External Validity</strong><span class="method-status">Partial</span></div><p></p></article><article class="method-audit-item method-missing"><div class="method-audit-head"><strong>Cost Evidence</strong><span class="method-status">missing</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Stakeholder Representation</strong><span class="method-status">Partial</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Implementation Evidence</strong><span class="method-status">Partial</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Equity Analysis</strong><span class="method-status">Partial</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Effect Size Reported</strong><span class="method-status">Partial</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Uncertainty Quantified</strong><span class="method-status">Partial</span></div><p></p></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>Comparator Clarity</strong><span class="method-status">Met</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Publication Bias Risk</strong><span class="method-status">Partial</span></div><p></p></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>Generalizability Claims</strong><span class="method-status">Met</span></div><p></p></article></div><div class="method-guard-wrap"><strong>Task vs learning guard: </strong><p class="method-guard">Teaching outcomes are not applicable; productivity is not a claim about student learning.</p></div></section><p>No data.</p><div class="evidence-tribunal" data-visual="evidence-tribunal-grid"><p class="tribunal-summary"><strong>Decision: </strong>Pilot · <strong>Confidence: </strong>Moderate</p><div class="tribunal-grid"><article class="tribunal-card supported"><header><span aria-hidden="true">✓</span><h3>Can claim</h3><span class="tribunal-count">4</span></header><ul><li><p>AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome. — E-001</p><div class="tribunal-evidence-refs"><code>E-001</code></div></li><li><p>ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support. — E-002</p><div class="tribunal-evidence-refs"><code>E-002</code></div></li><li><p>AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support. — E-003</p><div class="tribunal-evidence-refs"><code>E-003</code></div></li><li><p>Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits. — E-004</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li></ul></article><article class="tribunal-card uncertain"><header><span aria-hidden="true">?</span><h3>Cannot yet claim</h3><span class="tribunal-count">1</span></header><ul><li><p>Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set [no direct evidence].</p></li></ul></article><article class="tribunal-card contradicted"><header><span aria-hidden="true">×</span><h3>Contradicted claims</h3><span class="tribunal-count">0</span></header><ul><li>No data.</li></ul></article><article class="tribunal-card missing"><header><span aria-hidden="true">…</span><h3>Missing evidence</h3><span class="tribunal-count">1</span></header><ul><li><p>Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set [no direct evidence].</p></li></ul></article></div><details class="supporting-visual"><summary>EvidenceFlow Protocol</summary><svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="EvidenceFlow Protocol"><title>EvidenceFlow Protocol</title><desc>Research flow from framing, retrieval, fetch/verify, extraction, challenge, method audit and adjudication to applicability and intervention evaluation.</desc><defs><marker id="arr-en-full-workflow" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">EvidenceFlow Protocol</text><rect x="24" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="56.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="56.0" y="126.0">Frame</tspan></text><line x1="88" y1="126" x2="100" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="100" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="132.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="132.0" y="126.0">Retrieve</tspan></text><line x1="164" y1="126" x2="176" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="176" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="208.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="208.0" y="126.0">Fetch</tspan><tspan x="208.0" y="140.0">Verify</tspan></text><line x1="240" y1="126" x2="252" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="252" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="284.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="284.0" y="126.0">Extract</tspan></text><line x1="316" y1="126" x2="328" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="328" y="100" width="64" height="52" rx="8" fill="#5E8A6A"/><text x="360.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="360.0" y="126.0">Challenge</tspan></text><line x1="392" y1="126" x2="404" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="404" y="100" width="64" height="52" rx="8" fill="#5E8A6A"/><text x="436.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="436.0" y="126.0">Audit</tspan><tspan x="436.0" y="140.0">Method</tspan></text><line x1="468" y1="126" x2="480" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="480" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="512.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="512.0" y="126.0">Adjudicate</tspan></text><line x1="544" y1="126" x2="556" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="556" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="588.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="588.0" y="126.0">Applicability</tspan></text><line x1="620" y1="126" x2="632" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="632" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="664.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="664.0" y="126.0">Intervene</tspan><tspan x="664.0" y="140.0">Evaluate</tspan></text></svg></details><details class="supporting-visual"><summary>Tribunal infographic</summary><svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="Evidence Tribunal infographic"><title>Evidence Tribunal infographic</title><desc>Evidence IDs for claims that can and cannot be claimed, plus the recommended action badge; full claim text is in the tribunal cards below.</desc><defs><marker id="arr-en-full-tribunal" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">Evidence Tribunal</text><text x="24" y="180" font-size="13" font-weight="700" fill="#3A3833">Source of conflict</text><text x="24" y="202" font-size="11" fill="#8A867E">See tribunal cards below</text><rect x="24" y="216" width="180" height="28" rx="14" fill="#C99A4A"/><text x="114" y="235" text-anchor="middle" font-size="13" font-weight="700" fill="#fff">Pilot</text><text x="250" y="80" font-size="13" font-weight="700" fill="#5E8A6A">Can claim (4)</text><circle cx="254" cy="100" r="3" fill="#5E8A6A"/><text x="266" y="105" font-size="11" fill="#3A3833">E-001</text><circle cx="254" cy="122" r="3" fill="#5E8A6A"/><text x="266" y="127" font-size="11" fill="#3A3833">E-002</text><circle cx="254" cy="144" r="3" fill="#5E8A6A"/><text x="266" y="149" font-size="11" fill="#3A3833">E-003</text><circle cx="254" cy="166" r="3" fill="#5E8A6A"/><text x="266" y="171" font-size="11" fill="#3A3833">E-004</text><text x="490" y="80" font-size="13" font-weight="700" fill="#A85B53">Cannot claim (1)</text><circle cx="494" cy="100" r="3" fill="#A85B53"/><text x="506" y="105" font-size="11" fill="#3A3833">(none)</text></svg></details></div><div class="trace-chain" data-visual="trace-chain"><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-001</strong><p class="trace-claim-text">AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-001</code><span>Policy Effectiveness</span><span class="dir pos">Positive effect</span><span class="trace-relation">claim relation: Support</span><span>Q 9.0</span><span class="trace-arrow">→</span><a href="https://academic.oup.com/qje/article/140/2/889/7990658"><code>S-001</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-002</strong><p class="trace-claim-text">ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-002</code><span>Policy Effectiveness</span><span class="dir pos">Positive effect</span><span class="trace-relation">claim relation: Support</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf"><code>S-002</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-003</strong><p class="trace-claim-text">AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-003</code><span>Implementation Risk</span><span class="dir neg">Negative effect</span><span class="trace-relation">claim relation: Support</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838"><code>S-003</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-004</strong><p class="trace-claim-text">Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-004</code><span>Implementation Risk</span><span class="dir neg">Negative effect</span><span class="trace-relation">claim relation: Support</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://academic.oup.com/qje/article/140/2/889/7990658"><code>S-001</code></a></div></div></article></div><div id="chart-trace-en" class="chart-mount" aria-label="Claim-Evidence Trace"></div><p class="chart-interpretation"><strong>What this means: </strong>Every important claim must resolve to Evidence IDs and original sources.</p></div></section><section id="full-04-action-en" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>04 Applicability & Teaching Action</h2><p class="full-chapter-lead">Connect applicability, guardrails and teaching actions to specific evidence.</p></header><div class="full-chapter-body"><p><strong>Target population: </strong>Enterprise customer-support staff, stratified by tenure and baseline skill.</p>
|
|
2412
|
+
</div><div class="scope-grid"><article class="scope-card"><h3>Research question</h3><p class="scope-text">Should an enterprise customer-support team introduce a generative AI assistant?</p></article><article class="scope-card"><h3>Intervention plan</h3><p class="scope-text">Policy: Human-supervised customer-support assistant; Policy type: Institutional reform; Mechanism: Approved knowledge base + agent review</p></article><article class="scope-card"><h3>Comparison</h3><p class="scope-text">Existing support workflow without generative AI suggestions.</p></article><article class="scope-card"><h3>Outcome constructs</h3><p class="scope-text">Primary outcomes: Policy effectiveness, Implementation risk; Secondary outcomes: Cost-effectiveness, Equity, Feasibility</p></article><article class="scope-card"><h3>Research scope</h3><p class="scope-text">Time range: 2023–2026; Evidence Types: Quasi-experimental, Randomised controlled trial</p></article><article class="scope-card"><h3>Decision success condition</h3><p class="scope-text">Improve verified resolutions per paid staff hour while preserving service quality, privacy and staff autonomy.</p></article></div></div></section><section id="full-02-evidence-en" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>02 Key Evidence & Outcome Separation</h2><p class="full-chapter-lead">Place process outputs, real effects, durability and risk on one evidence map without conflating them.</p></header><div class="full-chapter-body"><div class="retrieval-grid"><article><h3>Inclusion criteria</h3><ul></ul></article><article><h3>Exclusion criteria</h3><ul></ul></article></div><div class="retrieval-coverage"><h3>Source coverage</h3><p><code>S-001</code> <code>S-002</code> <code>S-003</code></p><p class="retrieval-note">This report shows only retrieval metadata present in result; it does not fabricate PRISMA/funnel counts when none exist.</p></div><div class="outcome-separation" data-visual="outcome-separation"><div class="visual-heading"><h3>Outcome Separation · Process output ≠ policy effect</h3><p>Outcomes are adjudicated separately so pilot throughput or efficiency gains are not silently treated as real policy effects.</p><p class="semantic-note">Effect direction comes from evidence.effect_direction; supporting a claim does not imply a positive outcome.</p></div><div class="outcome-groups"><article class="outcome-group outcome-learning"><h3>Effect / target outcome</h3><ul><li><strong>Policy effectiveness</strong><span class="outcome-states"><span class="dir pos">Positive effect 2</span></span></li></ul></article><article class="outcome-group outcome-risk"><h3>Risk / implementation risk</h3><ul><li><strong>Implementation risk</strong><span class="outcome-states"><span class="dir neg">Negative effect 2</span></span></li></ul></article></div></div><div class='table-wrap outcome-table'><table class='data-table'><thead><tr><th>Outcome</th><th>Positive effect</th><th>Negative effect</th><th>Null effect</th><th>Evidence</th></tr></thead><tbody><tr><td><strong>Policy effectiveness</strong><span class='raw-tag' title='raw id'>policy_effectiveness</span></td><td class='num'>2</td><td class='num'>0</td><td class='num'>0</td><td><code>E-001</code> <code>E-002</code> </td></tr><tr><td><strong>Implementation risk</strong><span class='raw-tag' title='raw id'>implementation_risk</span></td><td class='num'>0</td><td class='num'>2</td><td class='num'>0</td><td><code>E-003</code> <code>E-004</code> </td></tr></tbody></table></div><div class='matrix-controls'><div class='matrix-tools'><input id='matrix-search-full-en' type='search' placeholder='Search evidence…' aria-label='Filter / search evidence'><select id='matrix-direction-full-en' aria-label='Filter by effect direction'><option value=''>All effects</option><option value='positive'>Positive effect</option><option value='negative'>Negative effect</option><option value='null'>Null effect</option></select><select id='matrix-outcome-full-en' aria-label='Filter by outcome type'><option value=''>All outcomes</option><option value='implementation_risk'>Implementation risk</option><option value='policy_effectiveness'>Policy effectiveness</option></select></div></div><div class='table-wrap matrix-wrap'><table id='evidence-matrix-full-en' class='data-table evidence-matrix'><thead><tr><th>ID</th><th>Outcome</th><th>Effect</th><th>Quality</th><th>Claim</th><th>Source</th></tr></thead><tbody><tr data-effect="positive" data-direction="positive" data-outcome="policy_effectiveness" data-search="e-001 st-001 sample-support-all generative ai at work ai assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome. customer-support agents positive s-001"><td><code>E-001</code></td><td><strong>Policy effectiveness</strong></td><td><span class="dir pos">Positive effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">9</span><span class="quality-meter" aria-hidden="true"><i style="width:90%"></i></span></div></td><td class="claim-cell"><p>AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>ST-001</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SAMPLE-support-all</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Generative AI at Work</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Generative AI at Work</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Quasi-experimental</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>Customer-support agents</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>5172</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>Chat handling time; separate throughput summary: about 15% more issues resolved per hour.</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Positive effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>One firm, one tool and a staggered nonrandom rollout; causal interpretation depends on identification assumptions.</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 1 · D2 Sample quality = 2 · D3 Measurement validity = 2 · D4 Temporal strength = 2 · D5 Directness = 2</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>9.0</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://academic.oup.com/qje/article/140/2/889/7990658</dd></div><div class="evidence-detail-row"><dt>Verifiable URL</dt><dd><a href="https://academic.oup.com/qje/article/140/2/889/7990658">https://academic.oup.com/qje/article/140/2/889/7990658</a></dd></div></dl></details></td><td><a class="source-link" href="https://academic.oup.com/qje/article/140/2/889/7990658" title="Generative AI at Work"><code>S-001</code></a></td></tr><tr data-effect="positive" data-direction="positive" data-outcome="policy_effectiveness" data-search="e-002 st-002 sample-writing experimental evidence on the productivity effects of generative artificial intelligence chatgpt reduced time on short professional writing tasks; this is indirect evidence for customer support. college-educated working professionals positive s-002"><td><code>E-002</code></td><td><strong>Policy effectiveness</strong></td><td><span class="dir pos">Positive effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">7</span><span class="quality-meter" aria-hidden="true"><i style="width:70%"></i></span></div></td><td class="claim-cell"><p>ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>ST-002</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SAMPLE-writing</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Experimental evidence on the productivity effects of generative artificial intelligence</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Experimental evidence on the productivity effects of generative artificial intelligence</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Randomized controlled trial</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>College-educated working professionals</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>453</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>Self-reported task time: approximately 40% lower; assessed writing quality approximately 18% higher is a separate measure.</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Positive effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>Indirect evidence, not directly generalizable: brief incentivized writing tasks did not demand precise factual accuracy or customer-specific context.</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 2 · D2 Sample quality = 2 · D3 Measurement validity = 1 · D4 Temporal strength = 1 · D5 Directness = 1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>7.0</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf</dd></div><div class="evidence-detail-row"><dt>Verifiable URL</dt><dd><a href="https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf">https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf</a></dd></div></dl></details></td><td><a class="source-link" href="https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf" title="Experimental evidence on the productivity effects of generative artificial intelligence"><code>S-002</code></a></td></tr><tr data-effect="negative" data-direction="negative" data-outcome="implementation_risk" data-search="e-003 st-003 sample-consulting-outside navigating the jagged technological frontier: field experimental evidence of the effects of artificial intelligence on knowledge worker productivity and quality ai can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support. bcg consultants in the outside-frontier experiment negative s-003"><td><code>E-003</code></td><td><strong>Implementation risk</strong></td><td><span class="dir neg">Negative effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>ST-003</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SAMPLE-consulting-outside</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2026</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Randomized controlled trial</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>BCG consultants in the outside-frontier experiment</dd></div><div class="evidence-detail-row"><dt>Sample size</dt><dd>373</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>Correct business recommendation: about 19 percentage points lower across AI arms; Table 7 has 373 participants, not 758.</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Negative effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>Indirect evidence, not directly generalizable: consultants solving an experimental business case, not live customer tickets.</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 2 · D2 Sample quality = 2 · D3 Measurement validity = 2 · D4 Temporal strength = 1 · D5 Directness = 1</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838</dd></div><div class="evidence-detail-row"><dt>Verifiable URL</dt><dd><a href="https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838">https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838</a></dd></div></dl></details></td><td><a class="source-link" href="https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838" title="Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality"><code>S-003</code></a></td></tr><tr data-effect="negative" data-direction="negative" data-outcome="implementation_risk" data-search="e-004 st-001 sample-support-all generative ai at work experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits. experienced and high-skill customer-support agents negative s-001"><td><code>E-004</code></td><td><strong>Implementation risk</strong></td><td><span class="dir neg">Negative effect</span><span class="relation-note">Claim relation: Support</span></td><td><div class="quality-cell"><span class="num">8</span><span class="quality-meter" aria-hidden="true"><i style="width:80%"></i></span></div></td><td class="claim-cell"><p>Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits.</p><details class="matrix-row-detail"><summary>View full evidence</summary><dl class="evidence-detail-grid"><div class="evidence-detail-row"><dt>Study ID</dt><dd>ST-001</dd></div><div class="evidence-detail-row"><dt>Sample ID</dt><dd>SAMPLE-support-all</dd></div><div class="evidence-detail-row"><dt>Study title</dt><dd>Generative AI at Work</dd></div><div class="evidence-detail-row"><dt>Source title</dt><dd>Generative AI at Work</dd></div><div class="evidence-detail-row"><dt>Year</dt><dd>2025</dd></div><div class="evidence-detail-row"><dt>Study design</dt><dd>Quasi-experimental</dd></div><div class="evidence-detail-row"><dt>Population</dt><dd>Experienced and high-skill customer-support agents</dd></div><div class="evidence-detail-row"><dt>Outcome measure</dt><dd>Small quality declines among the most experienced and highest-skilled support staff.</dd></div><div class="evidence-detail-row"><dt>Effect direction</dt><dd>Negative effect</dd></div><div class="evidence-detail-row"><dt>Relation to claim</dt><dd>Support</dd></div><div class="evidence-detail-row"><dt>Limitations</dt><dd>Same study as E-001; subgroup sample size was not extracted. This is not an independent replication.</dd></div><div class="evidence-detail-row"><dt>Quality dimensions</dt><dd>D1 Study design = 1 · D2 Sample quality = 1 · D3 Measurement validity = 2 · D4 Temporal strength = 2 · D5 Directness = 2</dd></div><div class="evidence-detail-row"><dt>Quality score</dt><dd>8.0</dd></div><div class="evidence-detail-row"><dt>Evidence status</dt><dd>Supported</dd></div><div class="evidence-detail-row"><dt>Full claim</dt><dd>Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits.</dd></div><div class="evidence-detail-row"><dt>Source location</dt><dd>https://academic.oup.com/qje/article/140/2/889/7990658</dd></div><div class="evidence-detail-row"><dt>Verifiable URL</dt><dd><a href="https://academic.oup.com/qje/article/140/2/889/7990658">https://academic.oup.com/qje/article/140/2/889/7990658</a></dd></div></dl></details></td><td><a class="source-link" href="https://academic.oup.com/qje/article/140/2/889/7990658" title="Generative AI at Work"><code>S-001</code></a></td></tr></tbody></table></div></div></section><section id="full-03-quality-en" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>03 Evidence Quality, Counterevidence & Method Audit</h2><p class="full-chapter-lead">Examine why evidence is credible, where it conflicts, and which conclusions require downgrading.</p></header><div class="full-chapter-body"><section class="method-review-group"><header class="method-review-title"><h3>Audit target: overall</h3><span class="method-verdict">Concern</span></header><div class="method-audit-grid"><article class="method-audit-item method-met"><div class="method-audit-head"><strong>Evidence level</strong><span class="method-status">Met</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Causal identification</strong><span class="method-status">Partial</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>External validity</strong><span class="method-status">Partial</span></div><p></p></article><article class="method-audit-item method-missing"><div class="method-audit-head"><strong>Cost evidence</strong><span class="method-status">missing</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Stakeholder representation</strong><span class="method-status">Partial</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Implementation evidence</strong><span class="method-status">Partial</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Equity analysis</strong><span class="method-status">Partial</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Effect size reported</strong><span class="method-status">Partial</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Uncertainty quantified</strong><span class="method-status">Partial</span></div><p></p></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>Comparator clarity</strong><span class="method-status">Met</span></div><p></p></article><article class="method-audit-item method-partial"><div class="method-audit-head"><strong>Publication-bias risk</strong><span class="method-status">Partial</span></div><p></p></article><article class="method-audit-item method-met"><div class="method-audit-head"><strong>Generalizability claims</strong><span class="method-status">Met</span></div><p></p></article></div><div class="method-guard-wrap"><strong>Output vs effect guard: </strong><p class="method-guard">Teaching outcomes are not applicable; productivity is not a claim about student learning.</p></div></section><p>No data.</p><div class="evidence-tribunal" data-visual="evidence-tribunal-grid"><p class="tribunal-summary"><strong>Decision: </strong>Pilot · <strong>Confidence: </strong>Moderate</p><div class="tribunal-grid"><article class="tribunal-card supported"><header><span aria-hidden="true">✓</span><h3>Can claim</h3><span class="tribunal-count">4</span></header><ul><li><p>AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome. — E-001</p><div class="tribunal-evidence-refs"><code>E-001</code></div></li><li><p>ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support. — E-002</p><div class="tribunal-evidence-refs"><code>E-002</code></div></li><li><p>AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support. — E-003</p><div class="tribunal-evidence-refs"><code>E-003</code></div></li><li><p>Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits. — E-004</p><div class="tribunal-evidence-refs"><code>E-004</code></div></li></ul></article><article class="tribunal-card uncertain"><header><span aria-hidden="true">?</span><h3>Cannot yet claim</h3><span class="tribunal-count">1</span></header><ul><li><p>Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set [no direct evidence].</p></li></ul></article><article class="tribunal-card contradicted"><header><span aria-hidden="true">×</span><h3>Contradicted claims</h3><span class="tribunal-count">0</span></header><ul><li>No data.</li></ul></article><article class="tribunal-card missing"><header><span aria-hidden="true">…</span><h3>Missing evidence</h3><span class="tribunal-count">1</span></header><ul><li><p>Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set [no direct evidence].</p></li></ul></article></div><details class="supporting-visual"><summary>EvidenceFlow Protocol</summary><svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="EvidenceFlow Protocol"><title>EvidenceFlow Protocol</title><desc>Research flow from framing, retrieval, fetch/verify, extraction, challenge, method audit and adjudication to applicability and intervention evaluation.</desc><defs><marker id="arr-en-full-workflow" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">EvidenceFlow Protocol</text><rect x="24" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="56.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="56.0" y="126.0">Frame</tspan></text><line x1="88" y1="126" x2="100" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="100" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="132.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="132.0" y="126.0">Retrieve</tspan></text><line x1="164" y1="126" x2="176" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="176" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="208.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="208.0" y="126.0">Fetch</tspan><tspan x="208.0" y="140.0">Verify</tspan></text><line x1="240" y1="126" x2="252" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="252" y="100" width="64" height="52" rx="8" fill="#B8694A"/><text x="284.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="284.0" y="126.0">Extract</tspan></text><line x1="316" y1="126" x2="328" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="328" y="100" width="64" height="52" rx="8" fill="#5E8A6A"/><text x="360.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="360.0" y="126.0">Challenge</tspan></text><line x1="392" y1="126" x2="404" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="404" y="100" width="64" height="52" rx="8" fill="#5E8A6A"/><text x="436.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="436.0" y="126.0">Audit</tspan><tspan x="436.0" y="140.0">Method</tspan></text><line x1="468" y1="126" x2="480" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="480" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="512.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="512.0" y="126.0">Adjudicate</tspan></text><line x1="544" y1="126" x2="556" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="556" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="588.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="588.0" y="126.0">Applicability</tspan></text><line x1="620" y1="126" x2="632" y2="126" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-workflow)"/><rect x="632" y="100" width="64" height="52" rx="8" fill="#C99A4A"/><text x="664.0" y="126.0" text-anchor="middle" fill="#FFFFFF" font-size="10" font-weight="600"><tspan x="664.0" y="126.0">Intervene</tspan><tspan x="664.0" y="140.0">Evaluate</tspan></text></svg></details><details class="supporting-visual"><summary>Tribunal infographic</summary><svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="Evidence Tribunal infographic"><title>Evidence Tribunal infographic</title><desc>Evidence IDs for claims that can and cannot be claimed, plus the recommended action badge; full claim text is in the tribunal cards below.</desc><defs><marker id="arr-en-full-tribunal" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">Evidence Tribunal</text><text x="24" y="180" font-size="13" font-weight="700" fill="#3A3833">Source of conflict</text><text x="24" y="202" font-size="11" fill="#8A867E">See tribunal cards below</text><rect x="24" y="216" width="180" height="28" rx="14" fill="#C99A4A"/><text x="114" y="235" text-anchor="middle" font-size="13" font-weight="700" fill="#fff">Pilot</text><text x="250" y="80" font-size="13" font-weight="700" fill="#5E8A6A">Can claim (4)</text><circle cx="254" cy="100" r="3" fill="#5E8A6A"/><text x="266" y="105" font-size="11" fill="#3A3833">E-001</text><circle cx="254" cy="122" r="3" fill="#5E8A6A"/><text x="266" y="127" font-size="11" fill="#3A3833">E-002</text><circle cx="254" cy="144" r="3" fill="#5E8A6A"/><text x="266" y="149" font-size="11" fill="#3A3833">E-003</text><circle cx="254" cy="166" r="3" fill="#5E8A6A"/><text x="266" y="171" font-size="11" fill="#3A3833">E-004</text><text x="490" y="80" font-size="13" font-weight="700" fill="#A85B53">Cannot claim (1)</text><circle cx="494" cy="100" r="3" fill="#A85B53"/><text x="506" y="105" font-size="11" fill="#3A3833">(none)</text></svg></details></div><div class="trace-chain" data-visual="trace-chain"><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-001</strong><p class="trace-claim-text">AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-001</code><span>Policy effectiveness</span><span class="dir pos">Positive effect</span><span class="trace-relation">claim relation: Support</span><span>Q 9.0</span><span class="trace-arrow">→</span><a href="https://academic.oup.com/qje/article/140/2/889/7990658"><code>S-001</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-002</strong><p class="trace-claim-text">ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-002</code><span>Policy effectiveness</span><span class="dir pos">Positive effect</span><span class="trace-relation">claim relation: Support</span><span>Q 7.0</span><span class="trace-arrow">→</span><a href="https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf"><code>S-002</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-003</strong><p class="trace-claim-text">AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-003</code><span>Implementation risk</span><span class="dir neg">Negative effect</span><span class="trace-relation">claim relation: Support</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838"><code>S-003</code></a></div></div></article><article class="trace-chain-card"><div class="trace-claim-node"><strong>C-004</strong><p class="trace-claim-text">Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits.</p></div><div class="trace-arrow">↓</div><div class="trace-evidence-list"><div class="trace-evidence-node"><code>E-004</code><span>Implementation risk</span><span class="dir neg">Negative effect</span><span class="trace-relation">claim relation: Support</span><span>Q 8.0</span><span class="trace-arrow">→</span><a href="https://academic.oup.com/qje/article/140/2/889/7990658"><code>S-001</code></a></div></div></article></div><div id="chart-trace-en" class="chart-mount" aria-label="Claim-Evidence Trace"></div><p class="chart-interpretation"><strong>What this means: </strong>Every important claim must resolve to Evidence IDs and original sources.</p></div></section><section id="full-04-action-en" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>04 Applicability & Policy Action</h2><p class="full-chapter-lead">Connect applicability, guardrails and policy actions to specific evidence.</p></header><div class="full-chapter-body"><p><strong>Target population: </strong>Enterprise customer-support staff, stratified by tenure and baseline skill.</p>
|
|
2413
2413
|
<p><strong>Target context: </strong>Human-supervised support using an approved knowledge base.</p>
|
|
2414
2414
|
<p><strong>Conditions: </strong></p><ul><li>Start with low-risk, knowledge-base-covered queues; agents inspect and edit every proposed reply.</li><li>Stratify by tenure and baseline skill; retain expert discretion and measure review workload.</li><li>Minimize and redact customer data, restrict access and retention, and review supplier data-use terms before a pilot.</li><li>Escalate unfamiliar, ambiguous, sensitive or high-stakes requests to qualified humans; no autonomous commitments.</li></ul><div class="boundary-block"><h3>Claims beyond the evidence boundary</h3><ul><li>Claiming universal gains or that autonomous deployment is safe exceeds the boundary: no included study measures privacy incidents, local net cost or subgroup service quality.</li></ul></div><div class="evidence-to-action" data-visual="evidence-to-action"><article class="action-node action-evidence"><span>Evidence</span><p class="action-node-text">AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome. — E-001</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-applicability"><span>Applicability</span><div class="action-node-text"><p>Start with low-risk, knowledge-base-covered queues; agents inspect and edit every proposed reply.; Stratify by tenure and baseline skill;…</p><details class="detail-expander"><summary>Expand full explanation</summary><div class="detail-body"><p>Start with low-risk, knowledge-base-covered queues; agents inspect and edit every proposed reply.; Stratify by tenure and baseline skill; retain expert discretion and measure review workload.; Minimize and redact customer data, restrict access and retention, and review supplier data-use terms before a pilot.; Escalate unfamiliar, ambiguous, sensitive or high-stakes requests to qualified humans; no autonomous commitments.</p></div></details></div></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-decision"><span>Decision</span><p class="action-node-text">Pilot</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-guardrails"><span>Guardrails</span><p class="action-node-text">Start with low-risk, knowledge-base-covered queues; agents inspect and edit every proposed reply.</p></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-stop"><span>Stop conditions</span><div class="action-node-text"><p>Pause immediately after a verified privacy leak or a serious unsafe or unauthorized customer commitment; investigate before resuming.;…</p><details class="detail-expander"><summary>Expand full explanation</summary><div class="detail-body"><p>Pause immediately after a verified privacy leak or a serious unsafe or unauthorized customer commitment; investigate before resuming.; Pause expansion if blinded quality review crosses a pre-agreed noninferiority margin, including within experience strata.</p></div></details></div></article><span class="flow-arrow" aria-hidden="true">→</span><article class="action-node action-evaluation"><span>Evaluation</span><div class="action-node-text"><p>Expand only if quality is noninferior, verified resolutions per paid hour improve, net cost is acceptable and no serious unresolved safety incident re…</p><details class="detail-expander"><summary>Expand full explanation</summary><div class="detail-body"><p>Expand only if quality is noninferior, verified resolutions per paid hour improve, net cost is acceptable and no serious unresolved safety incident remains; thresholds require local agreement.</p></div></details></div></article></div><p><strong>Target population: </strong>Enterprise customer-support staff, stratified by tenure and baseline skill. · <strong>Pilot duration: </strong>Proposed: two baseline weeks and six pilot weeks; not executed.</p>
|
|
2415
|
-
<p><strong>
|
|
2415
|
+
<p><strong>Intervention usage policy: </strong>Start with low-risk, knowledge-base-covered queues; agents inspect and edit every proposed reply.</p>
|
|
2416
2416
|
<h3>Stop conditions</h3><ul><li>Pause immediately after a verified privacy leak or a serious unsafe or unauthorized customer commitment; investigate before resuming.</li><li>Pause expansion if blinded quality review crosses a pre-agreed noninferiority margin, including within experience strata.</li></ul>
|
|
2417
|
-
<h3>Intervention timeline infographic</h3>
|
|
2418
|
-
<svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="
|
|
2417
|
+
<h3>Intervention plan timeline infographic</h3>
|
|
2418
|
+
<svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="Intervention plan timeline"><title>Intervention plan timeline</title><desc>Short phase names and activity counts; full intervention usage rules are in the phase blocks.</desc><defs><marker id="arr-en-full-intervention" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">Intervention plan timeline</text><rect x="24" y="100" width="160" height="96" rx="8" fill="#B8694A"/><text x="104.0" y="148.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="104.0" y="148.0">Phase 1</tspan></text></svg></div></section><section id="full-05-evaluation-en" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>05 Pilot, Evaluation & Stop Conditions</h2><p class="full-chapter-lead">Validate the pilot with independent effect measures and pre-specified stop conditions.</p></header><div class="full-chapter-body"><p><strong>Research question: </strong>Should an enterprise customer-support team introduce a generative AI assistant?</p>
|
|
2419
2419
|
<p><strong>Baseline: </strong>Record resolution rate, paid hours, repeat contacts, blinded quality and review costs before allocation.</p>
|
|
2420
|
-
<p><strong>
|
|
2420
|
+
<p><strong>Pilot end: </strong>Assess the same outcomes at pilot end; separately audit privacy and unsafe commitments.</p>
|
|
2421
2421
|
<p><strong>Success threshold: </strong>Expand only if quality is noninferior, verified resolutions per paid hour improve, net cost is acceptable and no serious unresolved safety incident remains; thresholds require local agreement.</p>
|
|
2422
2422
|
<p><strong>Analysis plan: </strong>Preregister an intention-to-treat comparison with team-clustered uncertainty and subgroup estimates. Set sample size and noninferiority margins from local baseline variance and operational priorities before enrollment.</p>
|
|
2423
2423
|
<h3>Evaluation design infographic</h3>
|
|
2424
|
-
<svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="Evaluation design flow"><title>Evaluation design flow</title><desc>Evaluation flow across baseline,
|
|
2424
|
+
<svg viewBox="0 0 720 260" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="Evaluation design flow"><title>Evaluation design flow</title><desc>Evaluation flow across baseline, pilot end, durability and scale-out; full metrics and analysis plan are in the evaluation section.</desc><defs><marker id="arr-en-full-evaluation" markerWidth="8" markerHeight="8" refX="6" refY="3" orient="auto"><path d="M0,0 L6,3 L0,6 Z" fill="#8A867E"/></marker></defs><rect width="720" height="260" fill="#FCFAF6" rx="12"/><text x="24" y="34" font-size="16" font-weight="700" fill="#3A3833">Evaluation design flow</text><rect x="30" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="105.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="105.0" y="138.0">Baseline</tspan></text><text x="105.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">pre-test</text><line x1="180" y1="138" x2="192" y2="138" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-evaluation)"/><rect x="192" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="267.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="267.0" y="138.0">Post test</tspan></text><text x="267.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">post-test</text><line x1="342" y1="138" x2="354" y2="138" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-evaluation)"/><rect x="354" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="429.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="429.0" y="138.0">Retention</tspan></text><text x="429.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">Durability</text><line x1="504" y1="138" x2="516" y2="138" stroke="#8A867E" stroke-width="2" marker-end="url(#arr-en-full-evaluation)"/><rect x="516" y="110" width="150" height="56" rx="8" fill="#5E8A6A"/><text x="591.0" y="138.0" text-anchor="middle" fill="#FFFFFF" font-size="12" font-weight="600"><tspan x="591.0" y="138.0">Transfer</tspan></text><text x="591.0" y="148" text-anchor="middle" font-size="9" fill="#fff" opacity="0.95">Scale-out</text></svg><div class="visual-suppressed"><strong>Benchmark visual suppressed</strong><p>result.json carries no benchmark.baselines, so this visual is omitted; see the standalone benchmark report.</p></div></div></section><section id="full-06-sources-en" class="full-chapter" data-full-chapter><header class="full-chapter-header"><h2>06 Sources, Traceability & Appendix</h2><p class="full-chapter-lead">Preserve original sources, URLs, evidence IDs and retrieval metadata for auditability.</p></header><div class="full-chapter-body"><h3>Source list</h3><div class='table-wrap'><table class='data-table source-table'><thead><tr><th>ID</th><th>Title</th><th>Year</th><th>Authority</th><th>Verifiable location</th></tr></thead><tbody><tr><td><code>S-001</code></td><td class='cell-main source-title-cell'>Generative AI at Work<details class="source-expander detail-expander"><summary>View source & provenance</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>Original title</dt><dd>Generative AI at Work</dd></div><div class="source-detail-row"><dt>Year</dt><dd>2025</dd></div><div class="source-detail-row"><dt>Authority</dt><dd>Tier 1 DOI-verified paper</dd></div><div class="source-detail-row"><dt>Fetch provider</dt><dd>builtin</dd></div><div class="source-detail-row"><dt>Fetch status</dt><dd>FETCH_VALID</dd></div><div class="source-detail-row"><dt>Verifiable link</dt><dd><a href="https://academic.oup.com/qje/article/140/2/889/7990658">https://academic.oup.com/qje/article/140/2/889/7990658</a></dd></div></dl></details></td><td>2025</td><td>Tier 1 DOI-verified paper</td><td class='cell-main'><a href='https://academic.oup.com/qje/article/140/2/889/7990658'>https://academic.oup.com/qje/article/140/2/889/7990658</a></td></tr><tr><td><code>S-002</code></td><td class='cell-main source-title-cell'>Experimental evidence on the productivity effects of generative artificial intelligence<details class="source-expander detail-expander"><summary>View source & provenance</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>Original title</dt><dd>Experimental evidence on the productivity effects of generative artificial intelligence</dd></div><div class="source-detail-row"><dt>Year</dt><dd>2023</dd></div><div class="source-detail-row"><dt>Authority</dt><dd>Tier 1 DOI-verified paper</dd></div><div class="source-detail-row"><dt>Fetch provider</dt><dd>builtin</dd></div><div class="source-detail-row"><dt>Fetch status</dt><dd>FETCH_VALID</dd></div><div class="source-detail-row"><dt>Verifiable link</dt><dd><a href="https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf">https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf</a></dd></div></dl></details></td><td>2023</td><td>Tier 1 DOI-verified paper</td><td class='cell-main'><a href='https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf'>https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf</a></td></tr><tr><td><code>S-003</code></td><td class='cell-main source-title-cell'>Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality<details class="source-expander detail-expander"><summary>View source & provenance</summary><dl class="source-detail-grid"><div class="source-detail-row"><dt>Original title</dt><dd>Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality</dd></div><div class="source-detail-row"><dt>Year</dt><dd>2026</dd></div><div class="source-detail-row"><dt>Authority</dt><dd>Tier 1 DOI-verified paper</dd></div><div class="source-detail-row"><dt>Fetch provider</dt><dd>builtin</dd></div><div class="source-detail-row"><dt>Fetch status</dt><dd>FETCH_VALID</dd></div><div class="source-detail-row"><dt>Verifiable link</dt><dd><a href="https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838">https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838</a></dd></div></dl></details></td><td>2026</td><td>Tier 1 DOI-verified paper</td><td class='cell-main'><a href='https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838'>https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838</a></td></tr></tbody></table></div><h3>Fetch provenance</h3><div class='table-wrap'><table class='data-table'><thead><tr><th>Source</th><th>Fetch method</th><th>Status</th><th>Fallback</th><th>Time</th></tr></thead><tbody><tr><td><code>S-001</code></td><td>builtin</td><td>FETCH_VALID</td><td></td><td></td></tr><tr><td><code>S-002</code></td><td>builtin</td><td>FETCH_VALID</td><td></td><td></td></tr><tr><td><code>S-003</code></td><td>builtin</td><td>FETCH_VALID</td><td></td><td></td></tr></tbody></table></div></div></section></main></div>
|
|
2425
2425
|
</div>
|
|
2426
2426
|
<footer class="report-footer"><p>EduEvidence Evidence Report · Schema PASS · Claim Binding PASS · Numeric Consistency PASS · Bilingual Structure PASS · Human Language PASS · False Precision PASS · Lieflat Data Bound PASS · Axis Distortion NOT_CHECKED · Colorblind Safe NOT_CHECKED · single-file offline · source: result.json</p></footer>
|
|
2427
2427
|
</div>
|
|
@@ -2668,10 +2668,10 @@ select:focus-visible,
|
|
|
2668
2668
|
window.eduevidenceCharts = window.eduevidenceCharts || {};
|
|
2669
2669
|
window.eduevidenceCharts[containerId] = chart;
|
|
2670
2670
|
}
|
|
2671
|
-
mountChart('chart-outcome-zh', {"chart_id": "outcome-evidence-overview", "purpose": "interactive_analysis", "engine": "echarts", "chart_type": "diverging_bar", "semantic_basis": "effect_direction", "title": "结果证据概览", "option": {"tooltip": {"trigger": "axis"}, "legend": {"data": ["正向效应", "负向效应", "零效应"]}, "grid": [{"left": 150, "right": 40, "top": 30, "height": "52%"}, {"left": 150, "right": 40, "top": "70%", "height": "18%"}], "xAxis": [{"type": "value", "gridIndex": 0, "minInterval": 1}, {"type": "value", "gridIndex": 1, "min": 0, "max": 2, "minInterval": 1}], "yAxis": [{"type": "category", "data": ["
|
|
2671
|
+
mountChart('chart-outcome-zh', {"chart_id": "outcome-evidence-overview", "purpose": "interactive_analysis", "engine": "echarts", "chart_type": "diverging_bar", "semantic_basis": "effect_direction", "title": "结果证据概览", "option": {"tooltip": {"trigger": "axis"}, "legend": {"data": ["正向效应", "负向效应", "零效应"]}, "grid": [{"left": 150, "right": 40, "top": 30, "height": "52%"}, {"left": 150, "right": 40, "top": "70%", "height": "18%"}], "xAxis": [{"type": "value", "gridIndex": 0, "minInterval": 1}, {"type": "value", "gridIndex": 1, "min": 0, "max": 2, "minInterval": 1}], "yAxis": [{"type": "category", "data": ["政策有效性", "实施风险"], "inverse": true, "gridIndex": 0}, {"type": "category", "data": ["政策有效性", "实施风险"], "inverse": true, "gridIndex": 1, "show": false}], "series": [{"name": "正向效应", "type": "bar", "data": [2, 0], "itemStyle": {"color": "#5E8A6A"}, "xAxisIndex": 0, "yAxisIndex": 0, "lane": "main"}, {"name": "负向效应", "type": "bar", "data": [0, -2], "itemStyle": {"color": "#A85B53"}, "xAxisIndex": 0, "yAxisIndex": 0, "lane": "main"}, {"name": "零效应", "type": "bar", "data": [0, 0], "itemStyle": {"color": "#C99A4A"}, "xAxisIndex": 1, "yAxisIndex": 1, "barWidth": 6, "lane": "neutral"}]}, "summary_text": "各 Outcome 的正向 / 负向 / 零效应证据条数对比。这里展示的是 effect_direction,不是“这条证据是否支持某个主张”。", "integrity": {"numbers_match_result": "NOT_CHECKED", "no_axis_distortion": "NOT_CHECKED", "no_false_precision": "NOT_CHECKED", "colorblind_safe": "NOT_CHECKED"}});
|
|
2672
2672
|
mountChart('chart-trace-zh', {"chart_id": "claim-evidence-trace", "purpose": "interactive_analysis", "engine": "echarts", "chart_type": "graph", "title": "主张-证据追溯", "option": {"tooltip": {"trigger": "item"}, "legend": {"data": ["Decision", "Claim", "Evidence", "Source"]}, "series": [{"type": "graph", "layout": "force", "roam": true, "draggable": true, "categories": [{"name": "Decision"}, {"name": "Claim"}, {"name": "Evidence"}, {"name": "Source"}], "data": [{"id": "decision", "name": "PILOT", "category": 0}, {"id": "claim-0", "name": "在有边界的客服场景中,AI 辅助可缩短会话处理时间;吞吐量是另一项指标。", "category": 1}, {"id": "E-001", "name": "E-001", "category": 2}, {"id": "S-001", "name": "S-001", "category": 3}, {"id": "claim-1", "name": "ChatGPT 缩短了短篇职业写作任务用时;对客服而言属于间接证据。", "category": 1}, {"id": "E-002", "name": "E-002", "category": 2}, {"id": "S-002", "name": "S-002", "category": 3}, {"id": "claim-2", "name": "在超出能力边界的任务中 AI 可能降低正确率;咨询实验对客服属于间接证据。", "category": 1}, {"id": "E-003", "name": "E-003", "category": 2}, {"id": "S-003", "name": "S-003", "category": 3}, {"id": "claim-3", "name": "须单独监测资深高技能客服的质量,不能假定所有岗位群体均受益。", "category": 1}, {"id": "E-004", "name": "E-004", "category": 2}], "links": [{"source": "decision", "target": "claim-0"}, {"source": "claim-0", "target": "E-001", "label": "support", "lineStyle": {"color": "#5E8A6A"}}, {"source": "E-001", "target": "S-001"}, {"source": "decision", "target": "claim-1"}, {"source": "claim-1", "target": "E-002", "label": "support", "lineStyle": {"color": "#5E8A6A"}}, {"source": "E-002", "target": "S-002"}, {"source": "decision", "target": "claim-2"}, {"source": "claim-2", "target": "E-003", "label": "support", "lineStyle": {"color": "#5E8A6A"}}, {"source": "E-003", "target": "S-003"}, {"source": "decision", "target": "claim-3"}, {"source": "claim-3", "target": "E-004", "label": "support", "lineStyle": {"color": "#5E8A6A"}}, {"source": "E-004", "target": "S-001"}], "label": {"show": true, "position": "right", "fontSize": 9}, "lineStyle": {"curveness": 0.15}}]}, "summary_text": "决策→结论→证据→来源 的可追溯图谱;点击节点可追踪支持/反驳路径。", "integrity": {"numbers_match_result": "NOT_CHECKED", "no_axis_distortion": "NOT_CHECKED", "no_false_precision": "NOT_CHECKED", "colorblind_safe": "NOT_CHECKED"}});
|
|
2673
2673
|
mountChart('chart-benchmark-zh', {"chart_id": "benchmark-panel", "purpose": "interactive_analysis", "engine": "echarts", "chart_type": "composite", "title": "基准测试:B0-B4", "option": {"tooltip": {"trigger": "axis"}, "legend": {"data": ["Citation Support", "Unsupported Rate", "Contradiction"]}, "grid": {"left": 60, "right": 40}, "xAxis": {"type": "category", "data": []}, "yAxis": {"type": "value", "max": 1, "min": 0}, "series": [{"name": "Citation Support", "type": "bar", "data": []}, {"name": "Unsupported Rate", "type": "bar", "data": []}, {"name": "Contradiction", "type": "bar", "data": []}]}, "cost_vs_quality": {"chart_id": "benchmark-quality-cost", "purpose": "interactive_analysis", "engine": "echarts", "chart_type": "scatter", "title": "质量 vs 成本", "option": {"tooltip": {"trigger": "item"}, "xAxis": {"type": "value", "name": "cost (USD)", "min": 0}, "yAxis": {"type": "value", "name": "citation support", "min": 0, "max": 1}, "series": [{"type": "scatter", "data": []}]}}, "summary_text": "B0-B4 基线在引用支持/无支撑率/反方发现上的对比及质量-成本散点。", "integrity": {"numbers_match_result": "NOT_CHECKED", "no_axis_distortion": "NOT_CHECKED", "no_false_precision": "NOT_CHECKED", "colorblind_safe": "NOT_CHECKED"}});
|
|
2674
|
-
mountChart('chart-outcome-en', {"chart_id": "outcome-evidence-overview", "purpose": "interactive_analysis", "engine": "echarts", "chart_type": "diverging_bar", "semantic_basis": "effect_direction", "title": "Outcome Evidence Overview", "option": {"tooltip": {"trigger": "axis"}, "legend": {"data": ["Positive effect", "Negative effect", "Null effect"]}, "grid": [{"left": 150, "right": 40, "top": 30, "height": "52%"}, {"left": 150, "right": 40, "top": "70%", "height": "18%"}], "xAxis": [{"type": "value", "gridIndex": 0, "minInterval": 1}, {"type": "value", "gridIndex": 1, "min": 0, "max": 2, "minInterval": 1}], "yAxis": [{"type": "category", "data": ["Policy
|
|
2674
|
+
mountChart('chart-outcome-en', {"chart_id": "outcome-evidence-overview", "purpose": "interactive_analysis", "engine": "echarts", "chart_type": "diverging_bar", "semantic_basis": "effect_direction", "title": "Outcome Evidence Overview", "option": {"tooltip": {"trigger": "axis"}, "legend": {"data": ["Positive effect", "Negative effect", "Null effect"]}, "grid": [{"left": 150, "right": 40, "top": 30, "height": "52%"}, {"left": 150, "right": 40, "top": "70%", "height": "18%"}], "xAxis": [{"type": "value", "gridIndex": 0, "minInterval": 1}, {"type": "value", "gridIndex": 1, "min": 0, "max": 2, "minInterval": 1}], "yAxis": [{"type": "category", "data": ["Policy effectiveness", "Implementation risk"], "inverse": true, "gridIndex": 0}, {"type": "category", "data": ["Policy effectiveness", "Implementation risk"], "inverse": true, "gridIndex": 1, "show": false}], "series": [{"name": "Positive effect", "type": "bar", "data": [2, 0], "itemStyle": {"color": "#5E8A6A"}, "xAxisIndex": 0, "yAxisIndex": 0, "lane": "main"}, {"name": "Negative effect", "type": "bar", "data": [0, -2], "itemStyle": {"color": "#A85B53"}, "xAxisIndex": 0, "yAxisIndex": 0, "lane": "main"}, {"name": "Null effect", "type": "bar", "data": [0, 0], "itemStyle": {"color": "#C99A4A"}, "xAxisIndex": 1, "yAxisIndex": 1, "barWidth": 6, "lane": "neutral"}]}, "summary_text": "Positive / negative / null effect-direction evidence counts per outcome. This visual encodes effect_direction, not whether evidence supports a claim.", "integrity": {"numbers_match_result": "NOT_CHECKED", "no_axis_distortion": "NOT_CHECKED", "no_false_precision": "NOT_CHECKED", "colorblind_safe": "NOT_CHECKED"}});
|
|
2675
2675
|
mountChart('chart-trace-en', {"chart_id": "claim-evidence-trace", "purpose": "interactive_analysis", "engine": "echarts", "chart_type": "graph", "title": "Claim-Evidence Trace", "option": {"tooltip": {"trigger": "item"}, "legend": {"data": ["Decision", "Claim", "Evidence", "Source"]}, "series": [{"type": "graph", "layout": "force", "roam": true, "draggable": true, "categories": [{"name": "Decision"}, {"name": "Claim"}, {"name": "Evidence"}, {"name": "Source"}], "data": [{"id": "decision", "name": "PILOT", "category": 0}, {"id": "claim-0", "name": "AI assistance can shorten customer chat ", "category": 1}, {"id": "E-001", "name": "E-001", "category": 2}, {"id": "S-001", "name": "S-001", "category": 3}, {"id": "claim-1", "name": "ChatGPT reduced time on short profession", "category": 1}, {"id": "E-002", "name": "E-002", "category": 2}, {"id": "S-002", "name": "S-002", "category": 3}, {"id": "claim-2", "name": "AI can reduce correctness on tasks outsi", "category": 1}, {"id": "E-003", "name": "E-003", "category": 2}, {"id": "S-003", "name": "S-003", "category": 3}, {"id": "claim-3", "name": "Experienced, high-skill support staff ne", "category": 1}, {"id": "E-004", "name": "E-004", "category": 2}], "links": [{"source": "decision", "target": "claim-0"}, {"source": "claim-0", "target": "E-001", "label": "support", "lineStyle": {"color": "#5E8A6A"}}, {"source": "E-001", "target": "S-001"}, {"source": "decision", "target": "claim-1"}, {"source": "claim-1", "target": "E-002", "label": "support", "lineStyle": {"color": "#5E8A6A"}}, {"source": "E-002", "target": "S-002"}, {"source": "decision", "target": "claim-2"}, {"source": "claim-2", "target": "E-003", "label": "support", "lineStyle": {"color": "#5E8A6A"}}, {"source": "E-003", "target": "S-003"}, {"source": "decision", "target": "claim-3"}, {"source": "claim-3", "target": "E-004", "label": "support", "lineStyle": {"color": "#5E8A6A"}}, {"source": "E-004", "target": "S-001"}], "label": {"show": true, "position": "right", "fontSize": 9}, "lineStyle": {"curveness": 0.15}}]}, "summary_text": "Traceable graph Decision → Claim → Evidence → Source; click nodes to follow support/contradict paths.", "integrity": {"numbers_match_result": "NOT_CHECKED", "no_axis_distortion": "NOT_CHECKED", "no_false_precision": "NOT_CHECKED", "colorblind_safe": "NOT_CHECKED"}});
|
|
2676
2676
|
mountChart('chart-benchmark-en', {"chart_id": "benchmark-panel", "purpose": "interactive_analysis", "engine": "echarts", "chart_type": "composite", "title": "Benchmark: B0-B4", "option": {"tooltip": {"trigger": "axis"}, "legend": {"data": ["Citation Support", "Unsupported Rate", "Contradiction"]}, "grid": {"left": 60, "right": 40}, "xAxis": {"type": "category", "data": []}, "yAxis": {"type": "value", "max": 1, "min": 0}, "series": [{"name": "Citation Support", "type": "bar", "data": []}, {"name": "Unsupported Rate", "type": "bar", "data": []}, {"name": "Contradiction", "type": "bar", "data": []}]}, "cost_vs_quality": {"chart_id": "benchmark-quality-cost", "purpose": "interactive_analysis", "engine": "echarts", "chart_type": "scatter", "title": "Quality vs Cost", "option": {"tooltip": {"trigger": "item"}, "xAxis": {"type": "value", "name": "cost (USD)", "min": 0}, "yAxis": {"type": "value", "name": "citation support", "min": 0, "max": 1}, "series": [{"type": "scatter", "data": []}]}}, "summary_text": "B0-B4 baselines compared on citation support, unsupported rate and contradiction discovery, plus the quality-cost scatter.", "integrity": {"numbers_match_result": "NOT_CHECKED", "no_axis_distortion": "NOT_CHECKED", "no_false_precision": "NOT_CHECKED", "colorblind_safe": "NOT_CHECKED"}});
|
|
2677
2677
|
})();
|