eduevidence 6.0.0 → 6.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (267) hide show
  1. package/CHANGELOG.md +395 -0
  2. package/CONTRIBUTING.md +105 -0
  3. package/README.md +113 -49
  4. package/README.zh-CN.md +39 -12
  5. package/SKILL.md +15 -5
  6. package/assets/readme/landing-tour.gif +0 -0
  7. package/assets/readme/studio-tour.gif +0 -0
  8. package/benchmarks/evidence-library.json +277 -1
  9. package/bin/eduevidence.js +2 -1
  10. package/docs/architecture.md +325 -46
  11. package/docs/demo-workplace-ai.md +1 -1
  12. package/docs/install-guide.md +1 -1
  13. package/docs/j-ev-experimental.md +250 -0
  14. package/docs/orchestration-role-model.md +1 -1
  15. package/docs/release-closeout/README.md +1 -1
  16. package/docs/reproducibility.md +138 -0
  17. package/docs/sciverse-api.md +125 -0
  18. package/domains/_neutral/copy/few_shots.json +21 -0
  19. package/domains/_neutral/copy/framing_lexicon.json +19 -0
  20. package/domains/_neutral/copy/module_labels.json +5 -0
  21. package/domains/_neutral/copy/module_labels_footer.json +102 -0
  22. package/domains/_neutral/copy/module_labels_modules.json +204 -0
  23. package/domains/_neutral/copy/module_labels_nav.json +126 -0
  24. package/domains/_neutral/copy/module_labels_summary.json +98 -0
  25. package/domains/_neutral/copy/module_labels_tables.json +164 -0
  26. package/domains/_neutral/copy/module_labels_v2.json +90 -0
  27. package/domains/_neutral/copy/risk_constructs.json +20 -0
  28. package/domains/_neutral/copy/section_titles.json +66 -0
  29. package/domains/_neutral/copy/terminology.json +11 -0
  30. package/domains/check_copy_packs.py +103 -0
  31. package/domains/education/copy/few_shots.json +22 -0
  32. package/domains/education/copy/framing_enums.json +167 -0
  33. package/domains/education/copy/framing_lexicon.json +166 -0
  34. package/domains/education/copy/module_labels.json +169 -0
  35. package/domains/education/copy/risk_constructs.json +48 -0
  36. package/domains/education/copy/section_titles.json +186 -0
  37. package/domains/education/copy/terminology.json +70 -0
  38. package/domains/education/manifest.json +1 -1
  39. package/domains/education/outcome_taxonomy.json +2 -2
  40. package/domains/manifest.json +1 -1
  41. package/domains/policy/copy/few_shots.json +22 -0
  42. package/domains/policy/copy/framing_enums.json +94 -0
  43. package/domains/policy/copy/framing_lexicon.json +174 -0
  44. package/domains/policy/copy/module_labels.json +168 -0
  45. package/domains/policy/copy/risk_constructs.json +33 -0
  46. package/domains/policy/copy/section_titles.json +186 -0
  47. package/domains/policy/copy/terminology.json +64 -0
  48. package/eduevidence_cli.py +10 -0
  49. package/engine/capabilities.py +57 -5
  50. package/engine/decision_policy.py +167 -0
  51. package/engine/evidence_graph.py +14 -10
  52. package/engine/gaps.py +42 -22
  53. package/engine/ids.py +2 -0
  54. package/engine/library.py +6 -2
  55. package/engine/library_builtin.py +7 -4
  56. package/engine/living.py +34 -4
  57. package/engine/migration.py +88 -3
  58. package/engine/orchestration.py +5 -5
  59. package/engine/paths.py +2 -0
  60. package/engine/pilot.py +34 -32
  61. package/engine/taxonomy.py +211 -0
  62. package/engine/tribunal.py +49 -43
  63. package/engine/versions.py +1 -1
  64. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1361 -147
  65. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +3 -3
  66. package/examples/ai-coding-assistant-evidence/citation_check.json +1 -1
  67. package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
  68. package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
  69. package/examples/ai-coding-assistant-evidence/report_spec.json +23 -12
  70. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +448 -128
  71. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +448 -128
  72. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +448 -128
  73. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +448 -128
  74. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +448 -128
  75. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1360 -146
  76. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1360 -146
  77. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1360 -146
  78. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1360 -146
  79. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1360 -146
  80. package/examples/ai-coding-assistant-evidence/result.json +13 -9
  81. package/examples/ai-coding-assistant-evidence/result.zh.json +45 -41
  82. package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
  83. package/examples/ai-coding-assistant-evidence/verdict.json +6 -2
  84. package/examples/spaced-retrieval-practice/EduEvidence_Report.html +2728 -0
  85. package/examples/spaced-retrieval-practice/applicability.json +14 -0
  86. package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
  87. package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
  88. package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
  89. package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
  90. package/examples/spaced-retrieval-practice/frame.json +58 -0
  91. package/examples/spaced-retrieval-practice/gate_report.json +101 -0
  92. package/examples/spaced-retrieval-practice/methodology.json +78 -0
  93. package/examples/spaced-retrieval-practice/report.html +2522 -0
  94. package/examples/spaced-retrieval-practice/report_spec.json +212 -0
  95. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
  96. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
  97. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
  98. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
  99. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
  100. package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
  101. package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
  102. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
  103. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
  104. package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
  105. package/examples/spaced-retrieval-practice/result.json +942 -0
  106. package/examples/spaced-retrieval-practice/result.zh.json +942 -0
  107. package/examples/spaced-retrieval-practice/skeptic.json +70 -0
  108. package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
  109. package/examples/spaced-retrieval-practice/verdict.json +93 -0
  110. package/examples/workplace-ai-assistant/EduEvidence_Report.html +2814 -0
  111. package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
  112. package/examples/workplace-ai-assistant/claims.jsonl +4 -4
  113. package/examples/workplace-ai-assistant/evidence.jsonl +4 -4
  114. package/examples/workplace-ai-assistant/evidence_graph.json +15 -15
  115. package/examples/workplace-ai-assistant/final_verdict.json +78 -0
  116. package/examples/workplace-ai-assistant/gate_report.json +101 -0
  117. package/examples/workplace-ai-assistant/report_spec.json +209 -40
  118. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +449 -119
  119. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +449 -119
  120. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +449 -119
  121. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +449 -119
  122. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +449 -119
  123. package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
  124. package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
  125. package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
  126. package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
  127. package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
  128. package/examples/workplace-ai-assistant/result.json +82 -20
  129. package/examples/workplace-ai-assistant/result.zh.json +82 -20
  130. package/examples/workplace-ai-assistant/skeptic.json +72 -0
  131. package/examples/workplace-ai-assistant/verdict.json +36 -10
  132. package/integrations/agent_mcp.py +2 -2
  133. package/integrations/jev/__init__.py +115 -0
  134. package/integrations/jev/approval.py +212 -0
  135. package/integrations/jev/cli.py +84 -0
  136. package/integrations/jev/config.py +112 -0
  137. package/integrations/jev/gateway.py +128 -0
  138. package/integrations/jev/modes.py +38 -0
  139. package/integrations/jev/tools_classify.py +88 -0
  140. package/integrations/jev/tools_extract.py +111 -0
  141. package/integrations/jev/tools_rerank.py +71 -0
  142. package/integrations/jev/tools_screen.py +87 -0
  143. package/integrations/jev/tools_verify.py +95 -0
  144. package/integrations/jev_mcp.py +22 -0
  145. package/integrations/semantic_decide.py +286 -0
  146. package/integrations/semdecide_cli.py +55 -0
  147. package/package.json +19 -2
  148. package/pyproject.toml +4 -3
  149. package/references/report-copy-style.md +107 -0
  150. package/references/retrieval-compliance.md +75 -0
  151. package/references/retrieval-protocol.md +20 -0
  152. package/retrieval/audit.py +27 -3
  153. package/retrieval/fetch.py +96 -0
  154. package/retrieval/sciverse.py +398 -0
  155. package/retrieval/search.py +47 -7
  156. package/schemas/applicability.schema.json +94 -0
  157. package/schemas/chart-spec.schema.json +10 -3
  158. package/schemas/evidence.schema.json +316 -43
  159. package/schemas/fetch-result.schema.json +2 -1
  160. package/schemas/report-result.schema.json +3 -3
  161. package/schemas/report-spec.schema.json +98 -100
  162. package/schemas/skeptic.schema.json +86 -0
  163. package/schemas/source.schema.json +21 -2
  164. package/schemas/v2/decision-snapshot.schema.json +20 -9
  165. package/schemas/v2/finding.schema.json +5 -1
  166. package/schemas/v2/intake.schema.json +191 -0
  167. package/schemas/v2/methodology-audit.schema.json +5 -1
  168. package/schemas/v2/outcome.schema.json +28 -5
  169. package/schemas/v2/study.schema.json +5 -1
  170. package/schemas/vNext/autoevolve-session.schema.json +34 -1
  171. package/schemas/vNext/eval-snapshot.schema.json +77 -1
  172. package/schemas/vNext/execution-plan.schema.json +50 -1
  173. package/schemas/vNext/gap-priority.schema.json +54 -1
  174. package/schemas/vNext/negative-search-record.schema.json +68 -1
  175. package/schemas/vNext/research-iteration.schema.json +87 -1
  176. package/schemas/vNext/research-strategy.schema.json +62 -1
  177. package/schemas/vNext/skill-experiment.schema.json +90 -1
  178. package/schemas/vNext/task-spec.schema.json +156 -1
  179. package/schemas/vNext/worker-result.schema.json +60 -1
  180. package/schemas/verdict.schema.json +164 -28
  181. package/scripts/build_evidence_library.py +15 -5
  182. package/scripts/build_report_variants.py +18 -2
  183. package/scripts/build_result.py +74 -9
  184. package/scripts/check_package_parity.py +85 -0
  185. package/scripts/check_protocol_alignment.py +375 -0
  186. package/scripts/check_versioned_schemas.py +254 -0
  187. package/scripts/claim_audit.py +13 -8
  188. package/scripts/compute_confidence.py +10 -0
  189. package/scripts/dashboard_server.py +13 -2
  190. package/scripts/did_regression.py +12 -2
  191. package/scripts/evidence_score.py +5 -2
  192. package/scripts/intake/__init__.py +31 -0
  193. package/scripts/intake/__main__.py +18 -0
  194. package/scripts/intake/background.py +78 -0
  195. package/scripts/intake/browser.py +79 -0
  196. package/scripts/intake/cli.py +57 -0
  197. package/scripts/intake/constants.py +57 -0
  198. package/scripts/intake/depth.py +53 -0
  199. package/scripts/intake/enhancements.py +106 -0
  200. package/scripts/intake/hooks.py +90 -0
  201. package/scripts/intake/prefs.py +76 -0
  202. package/scripts/intake/prompts.py +85 -0
  203. package/scripts/intake/session.py +152 -0
  204. package/scripts/lint_file_layers.py +126 -0
  205. package/scripts/orchestrator.py +187 -40
  206. package/scripts/pre_verdict_gate.py +241 -29
  207. package/scripts/quickstart.py +18 -2
  208. package/scripts/run_workspace.py +7 -1
  209. package/scripts/skill_lint.py +11 -1
  210. package/scripts/skill_payload.py +6 -3
  211. package/scripts/test_adversarial_empirical.py +96 -25
  212. package/scripts/validate_schema.py +31 -1
  213. package/skill/agents/evaluation-designer.md +20 -4
  214. package/skill/agents/evidence-analyst.md +19 -3
  215. package/skill/agents/evidence-judge.md +98 -8
  216. package/skill/agents/evidence-retriever.md +20 -3
  217. package/skill/agents/intervention-designer.md +20 -4
  218. package/skill/agents/method-reviewer.md +18 -2
  219. package/skill/agents/{education-planner.md → research-planner.md} +19 -3
  220. package/skill/agents/skeptic.md +18 -2
  221. package/skill/roles/registry.yaml +11 -11
  222. package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
  223. package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
  224. package/skill/sub-skills/data-analysis/SKILL.md +34 -15
  225. package/skill/sub-skills/ethics-review/SKILL.md +33 -10
  226. package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
  227. package/skill/sub-skills/evidence-review/SKILL.md +31 -12
  228. package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
  229. package/skill/sub-skills/literature-review/SKILL.md +35 -14
  230. package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
  231. package/skill/sub-skills/report-generation/SKILL.md +28 -0
  232. package/skill/sub-skills/research-planning/SKILL.md +41 -14
  233. package/skill/sub-skills/study-design/SKILL.md +30 -9
  234. package/skill/task-briefs/adjudicate.md +32 -7
  235. package/skill/task-briefs/applicability.md +37 -2
  236. package/skill/task-briefs/audit.md +32 -7
  237. package/skill/task-briefs/challenge.md +34 -5
  238. package/skill/task-briefs/evaluate.md +30 -5
  239. package/skill/task-briefs/extract.md +31 -8
  240. package/skill/task-briefs/frame.md +39 -10
  241. package/skill/task-briefs/intervene.md +32 -6
  242. package/skill/task-briefs/present.md +32 -8
  243. package/skill/task-briefs/projection.md +36 -2
  244. package/skill/task-briefs/retrieve.md +36 -6
  245. package/skill/workflows/decision-and-pilot.md +76 -1
  246. package/skill/workflows/evaluate-and-update.md +83 -0
  247. package/skill/workflows/evidence-review.md +104 -0
  248. package/skill/workflows/experimental-jev.md +170 -0
  249. package/skill/workflows/intake.md +120 -0
  250. package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
  251. package/visualization/eduevidence-report/scripts/build_infographics.py +37 -15
  252. package/visualization/eduevidence-report/scripts/build_report.py +435 -575
  253. package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
  254. package/visualization/eduevidence-report/scripts/lieflat_engine.py +349 -38
  255. package/visualization/eduevidence-report/scripts/report_copy_pack.py +296 -0
  256. package/visualization/eduevidence-report/scripts/report_copy_policy_guard.py +47 -0
  257. package/visualization/eduevidence-report/scripts/zh_labels.py +141 -1
  258. package/web/architecture.html +14885 -0
  259. package/web/studio/assets/index-B8tkF44Q.css +1 -0
  260. package/web/studio/index.html +2 -2
  261. package/scripts/build_esl_artifacts.py +0 -1921
  262. package/scripts/build_killer_demo.py +0 -295
  263. package/scripts/enrich_projects_human_and_lieflat.py +0 -315
  264. package/scripts/generate_new_projects.py +0 -686
  265. package/scripts/sync_killer_demo_report.py +0 -270
  266. package/web/studio/assets/index-CzXocaGv.css +0 -1
  267. /package/web/studio/assets/{index-pa7jD7n4.js → index-CQ6Keoyc.js} +0 -0
@@ -0,0 +1,15 @@
1
+ {
2
+ "result_sha256": "c26e7e1d60d3cf94bcba29233fbda453626aed47f77ea35b425c2535185316e7",
3
+ "result_zh_sha256": "65376ba253660d3fb6909f8d292ce1bd00ef92478dfa771f59ee95e9a2fc776b",
4
+ "renderer_version": "1.0.0",
5
+ "git_commit": "56f5dc3edc6e805c2614767208a83009b0f0a097",
6
+ "evidence_count": 4,
7
+ "source_count": 3,
8
+ "themes": [
9
+ "claude",
10
+ "academic",
11
+ "datalab",
12
+ "datalab-dark",
13
+ "presentation"
14
+ ]
15
+ }
@@ -1,4 +1,4 @@
1
- {"claim_id": "C-001", "claim": "AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.", "outcome_type": "completion_time", "evidence_ids": ["E-001"], "status": "SUPPORTED", "pooled_effect_g": null}
2
- {"claim_id": "C-002", "claim": "ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.", "outcome_type": "completion_time", "evidence_ids": ["E-002"], "status": "SUPPORTED", "pooled_effect_g": null}
3
- {"claim_id": "C-003", "claim": "AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.", "outcome_type": "accuracy", "evidence_ids": ["E-003"], "status": "SUPPORTED", "pooled_effect_g": null}
4
- {"claim_id": "C-004", "claim": "Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits.", "outcome_type": "accuracy", "evidence_ids": ["E-004"], "status": "SUPPORTED", "pooled_effect_g": null}
1
+ {"claim_id": "C-001", "claim": "AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.", "outcome_type": "policy_effectiveness", "evidence_ids": ["E-001"], "status": "SUPPORTED", "pooled_effect_g": null}
2
+ {"claim_id": "C-002", "claim": "ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.", "outcome_type": "policy_effectiveness", "evidence_ids": ["E-002"], "status": "SUPPORTED", "pooled_effect_g": null}
3
+ {"claim_id": "C-003", "claim": "AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.", "outcome_type": "implementation_risk", "evidence_ids": ["E-003"], "status": "SUPPORTED", "pooled_effect_g": null}
4
+ {"claim_id": "C-004", "claim": "Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits.", "outcome_type": "implementation_risk", "evidence_ids": ["E-004"], "status": "SUPPORTED", "pooled_effect_g": null}
@@ -1,4 +1,4 @@
1
- {"evidence_id": "E-001", "source_id": "S-001", "study_id": "ST-001", "sample_id": "SAMPLE-support-all", "claim_id": "C-001", "title": "Generative AI at Work", "year": 2025, "study_type": "quasi_experimental", "population": "Customer-support agents", "sample_size": 5172, "outcome_type": "completion_time", "outcome_measure": "Chat handling time; separate throughput summary: about 15% more issues resolved per hour.", "claim": "AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.", "direction": "support", "relation_to_claim": "support", "effect_direction": "positive", "decision_relation": "conditional", "source_location": "https://academic.oup.com/qje/article/140/2/889/7990658", "limitations": ["One firm, one tool and a staggered nonrandom rollout; causal interpretation depends on identification assumptions."], "status": "SUPPORTED", "extensions": {"domain": "policy", "policy_outcome": "policy_effectiveness", "directness": "direct", "raw_result": {"metric": "issues_resolved_per_hour_relative_change", "value": 15, "unit": "percent", "role": "separate_throughput_measure_not_completion_time", "ci_lower": null, "ci_upper": null, "p_value": null, "uncertainty_status": "not_extracted_for_this_summary_estimand"}, "standardized_effect": null, "study_total_n": 5172, "note": "Row-specific sample; study_total_n must not replace the analyzed sample. E-004 shares the S-001 cohort and is not separately counted."}}
2
- {"evidence_id": "E-002", "source_id": "S-002", "study_id": "ST-002", "sample_id": "SAMPLE-writing", "claim_id": "C-002", "title": "Experimental evidence on the productivity effects of generative artificial intelligence", "year": 2023, "study_type": "rct", "population": "College-educated working professionals", "sample_size": 453, "outcome_type": "completion_time", "outcome_measure": "Self-reported task time: approximately 40% lower; assessed writing quality approximately 18% higher is a separate measure.", "claim": "ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.", "direction": "support", "relation_to_claim": "support", "effect_direction": "positive", "decision_relation": "conditional", "source_location": "https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf", "limitations": ["Indirect evidence, not directly generalizable: brief incentivized writing tasks did not demand precise factual accuracy or customer-specific context."], "status": "SUPPORTED", "extensions": {"domain": "policy", "policy_outcome": "policy_effectiveness", "directness": "indirect", "raw_result": {"metric": "task_time_relative_change", "value": -40, "unit": "percent", "ci_lower": null, "ci_upper": null, "p_value": null, "uncertainty_status": "not_extracted_for_this_summary_estimand"}, "standardized_effect": null, "study_total_n": 453, "note": "Row-specific sample; study_total_n must not replace the analyzed sample. E-004 shares the S-001 cohort and is not separately counted."}}
3
- {"evidence_id": "E-003", "source_id": "S-003", "study_id": "ST-003", "sample_id": "SAMPLE-consulting-outside", "claim_id": "C-003", "title": "Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality", "year": 2026, "study_type": "rct", "population": "BCG consultants in the outside-frontier experiment", "sample_size": 373, "outcome_type": "accuracy", "outcome_measure": "Correct business recommendation: about 19 percentage points lower across AI arms; Table 7 has 373 participants, not 758.", "claim": "AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.", "direction": "support", "relation_to_claim": "support", "effect_direction": "negative", "decision_relation": "conditional", "source_location": "https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838", "limitations": ["Indirect evidence, not directly generalizable: consultants solving an experimental business case, not live customer tickets."], "status": "SUPPORTED", "extensions": {"domain": "policy", "policy_outcome": "implementation_risk", "directness": "indirect", "raw_result": {"metric": "correctness_absolute_change", "value": -19, "unit": "percentage_points", "ci_lower": null, "ci_upper": null, "p_value": null, "uncertainty_status": "not_extracted_for_this_summary_estimand"}, "standardized_effect": null, "study_total_n": 758, "note": "Row-specific sample; study_total_n must not replace the analyzed sample. E-004 shares the S-001 cohort and is not separately counted."}}
4
- {"evidence_id": "E-004", "source_id": "S-001", "study_id": "ST-001", "sample_id": "SAMPLE-support-all", "claim_id": "C-004", "title": "Generative AI at Work", "year": 2025, "study_type": "quasi_experimental", "population": "Experienced and high-skill customer-support agents", "sample_size": null, "outcome_type": "accuracy", "outcome_measure": "Small quality declines among the most experienced and highest-skilled support staff.", "claim": "Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits.", "direction": "support", "relation_to_claim": "support", "effect_direction": "negative", "decision_relation": "conditional", "source_location": "https://academic.oup.com/qje/article/140/2/889/7990658", "limitations": ["Same study as E-001; subgroup sample size was not extracted. This is not an independent replication."], "status": "SUPPORTED", "extensions": {"domain": "policy", "policy_outcome": "implementation_risk", "directness": "direct", "raw_result": {"metric": "experienced_staff_quality", "value": null, "unit": "not_extracted", "ci_lower": null, "ci_upper": null, "p_value": null, "uncertainty_status": "not_extracted_for_this_summary_estimand"}, "standardized_effect": null, "study_total_n": 5172, "note": "Row-specific sample; study_total_n must not replace the analyzed sample. E-004 shares the S-001 cohort and is not separately counted."}}
1
+ {"evidence_id": "E-001", "source_id": "S-001", "study_id": "ST-001", "sample_id": "SAMPLE-support-all", "claim_id": "C-001", "title": "Generative AI at Work", "year": 2025, "study_type": "quasi_experimental", "population": "Customer-support agents", "sample_size": 5172, "outcome_type": "policy_effectiveness", "outcome_measure": "Chat handling time; separate throughput summary: about 15% more issues resolved per hour.", "claim": "AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.", "direction": "support", "relation_to_claim": "support", "effect_direction": "positive", "decision_relation": "conditional", "source_location": "https://academic.oup.com/qje/article/140/2/889/7990658", "limitations": ["One firm, one tool and a staggered nonrandom rollout; causal interpretation depends on identification assumptions."], "status": "SUPPORTED", "quality_dimensions": {"D1_study_design": 1, "D2_sample_quality": 2, "D3_measurement_validity": 2, "D4_temporal_strength": 2, "D5_directness": 2}, "quality_score": 9.0, "extensions": {"domain": "policy", "policy_outcome": "policy_effectiveness", "teaching_neutral_outcome_token": "completion_time", "directness": "direct", "raw_result": {"metric": "issues_resolved_per_hour_relative_change", "value": 15, "unit": "percent", "role": "separate_throughput_measure_not_completion_time", "ci_lower": null, "ci_upper": null, "p_value": null, "uncertainty_status": "not_extracted_for_this_summary_estimand"}, "standardized_effect": null, "study_total_n": 5172, "note": "Row-specific sample; study_total_n must not replace the analyzed sample. E-004 shares the S-001 cohort and is not separately counted."}}
2
+ {"evidence_id": "E-002", "source_id": "S-002", "study_id": "ST-002", "sample_id": "SAMPLE-writing", "claim_id": "C-002", "title": "Experimental evidence on the productivity effects of generative artificial intelligence", "year": 2023, "study_type": "rct", "population": "College-educated working professionals", "sample_size": 453, "outcome_type": "policy_effectiveness", "outcome_measure": "Self-reported task time: approximately 40% lower; assessed writing quality approximately 18% higher is a separate measure.", "claim": "ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.", "direction": "support", "relation_to_claim": "support", "effect_direction": "positive", "decision_relation": "conditional", "source_location": "https://shakkednoy.com/Noy%20Zhang%20NBER%20SI.pdf", "limitations": ["Indirect evidence, not directly generalizable: brief incentivized writing tasks did not demand precise factual accuracy or customer-specific context."], "status": "SUPPORTED", "quality_dimensions": {"D1_study_design": 2, "D2_sample_quality": 2, "D3_measurement_validity": 1, "D4_temporal_strength": 1, "D5_directness": 1}, "quality_score": 7.0, "extensions": {"domain": "policy", "policy_outcome": "policy_effectiveness", "teaching_neutral_outcome_token": "completion_time", "directness": "indirect", "raw_result": {"metric": "task_time_relative_change", "value": -40, "unit": "percent", "ci_lower": null, "ci_upper": null, "p_value": null, "uncertainty_status": "not_extracted_for_this_summary_estimand"}, "standardized_effect": null, "study_total_n": 453, "note": "Row-specific sample; study_total_n must not replace the analyzed sample. E-004 shares the S-001 cohort and is not separately counted."}}
3
+ {"evidence_id": "E-003", "source_id": "S-003", "study_id": "ST-003", "sample_id": "SAMPLE-consulting-outside", "claim_id": "C-003", "title": "Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of Artificial Intelligence on Knowledge Worker Productivity and Quality", "year": 2026, "study_type": "rct", "population": "BCG consultants in the outside-frontier experiment", "sample_size": 373, "outcome_type": "implementation_risk", "outcome_measure": "Correct business recommendation: about 19 percentage points lower across AI arms; Table 7 has 373 participants, not 758.", "claim": "AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.", "direction": "support", "relation_to_claim": "support", "effect_direction": "negative", "decision_relation": "conditional", "source_location": "https://pubsonline.informs.org/doi/10.1287/orsc.2025.21838", "limitations": ["Indirect evidence, not directly generalizable: consultants solving an experimental business case, not live customer tickets."], "status": "SUPPORTED", "quality_dimensions": {"D1_study_design": 2, "D2_sample_quality": 2, "D3_measurement_validity": 2, "D4_temporal_strength": 1, "D5_directness": 1}, "quality_score": 8.0, "extensions": {"domain": "policy", "policy_outcome": "implementation_risk", "teaching_neutral_outcome_token": "accuracy", "directness": "indirect", "raw_result": {"metric": "correctness_absolute_change", "value": -19, "unit": "percentage_points", "ci_lower": null, "ci_upper": null, "p_value": null, "uncertainty_status": "not_extracted_for_this_summary_estimand"}, "standardized_effect": null, "study_total_n": 758, "note": "Row-specific sample; study_total_n must not replace the analyzed sample. E-004 shares the S-001 cohort and is not separately counted."}}
4
+ {"evidence_id": "E-004", "source_id": "S-001", "study_id": "ST-001", "sample_id": "SAMPLE-support-all", "claim_id": "C-004", "title": "Generative AI at Work", "year": 2025, "study_type": "quasi_experimental", "population": "Experienced and high-skill customer-support agents", "sample_size": null, "outcome_type": "implementation_risk", "outcome_measure": "Small quality declines among the most experienced and highest-skilled support staff.", "claim": "Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits.", "direction": "support", "relation_to_claim": "support", "effect_direction": "negative", "decision_relation": "conditional", "source_location": "https://academic.oup.com/qje/article/140/2/889/7990658", "limitations": ["Same study as E-001; subgroup sample size was not extracted. This is not an independent replication."], "status": "SUPPORTED", "quality_dimensions": {"D1_study_design": 1, "D2_sample_quality": 1, "D3_measurement_validity": 2, "D4_temporal_strength": 2, "D5_directness": 2}, "quality_score": 8.0, "extensions": {"domain": "policy", "policy_outcome": "implementation_risk", "teaching_neutral_outcome_token": "accuracy", "directness": "direct", "raw_result": {"metric": "experienced_staff_quality", "value": null, "unit": "not_extracted", "ci_lower": null, "ci_upper": null, "p_value": null, "uncertainty_status": "not_extracted_for_this_summary_estimand"}, "standardized_effect": null, "study_total_n": 5172, "note": "Row-specific sample; study_total_n must not replace the analyzed sample. E-004 shares the S-001 cohort and is not separately counted."}}
@@ -73,7 +73,7 @@
73
73
  "outcome_metric": "Chat handling time; separate throughput summary: about 15% more issues resolved per hour.",
74
74
  "outcome_dimension": "Policy",
75
75
  "claim_id": "C-001",
76
- "outcome_id": "completion_time",
76
+ "outcome_id": "policy_effectiveness",
77
77
  "effect_size": {
78
78
  "metric": "not_standardized",
79
79
  "value": null,
@@ -97,7 +97,7 @@
97
97
  "outcome_metric": "Self-reported task time: approximately 40% lower; assessed writing quality approximately 18% higher is a separate measure.",
98
98
  "outcome_dimension": "Policy",
99
99
  "claim_id": "C-002",
100
- "outcome_id": "completion_time",
100
+ "outcome_id": "policy_effectiveness",
101
101
  "effect_size": {
102
102
  "metric": "not_standardized",
103
103
  "value": null,
@@ -121,7 +121,7 @@
121
121
  "outcome_metric": "Correct business recommendation: about 19 percentage points lower across AI arms; Table 7 has 373 participants, not 758.",
122
122
  "outcome_dimension": "Policy",
123
123
  "claim_id": "C-003",
124
- "outcome_id": "accuracy",
124
+ "outcome_id": "implementation_risk",
125
125
  "effect_size": {
126
126
  "metric": "not_standardized",
127
127
  "value": null,
@@ -145,7 +145,7 @@
145
145
  "outcome_metric": "Small quality declines among the most experienced and highest-skilled support staff.",
146
146
  "outcome_dimension": "Policy",
147
147
  "claim_id": "C-004",
148
- "outcome_id": "accuracy",
148
+ "outcome_id": "implementation_risk",
149
149
  "effect_size": {
150
150
  "metric": "not_standardized",
151
151
  "value": null,
@@ -165,16 +165,16 @@
165
165
  }
166
166
  },
167
167
  "outcomes": {
168
- "completion_time": {
169
- "outcome_id": "completion_time",
170
- "name": "completion_time",
168
+ "policy_effectiveness": {
169
+ "outcome_id": "policy_effectiveness",
170
+ "name": "policy_effectiveness",
171
171
  "dimension": "Policy",
172
172
  "category": "Policy",
173
173
  "description": ""
174
174
  },
175
- "accuracy": {
176
- "outcome_id": "accuracy",
177
- "name": "accuracy",
175
+ "implementation_risk": {
176
+ "outcome_id": "implementation_risk",
177
+ "name": "implementation_risk",
178
178
  "dimension": "Policy",
179
179
  "category": "Policy",
180
180
  "description": ""
@@ -235,7 +235,7 @@
235
235
  "G-001": {
236
236
  "gap_id": "G-001",
237
237
  "gap_type": "Local applicability and safety",
238
- "description": "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set.",
238
+ "description": "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set [no direct evidence].",
239
239
  "target_outcome": "",
240
240
  "existing_evidence_summary": "E-001 through E-004",
241
241
  "recommended_trial_design": "Preregister an intention-to-treat comparison with team-clustered uncertainty and subgroup estimates. Set sample size and noninferiority margins from local baseline variance and operational priorities before enrollment."
@@ -316,7 +316,7 @@
316
316
  },
317
317
  {
318
318
  "source_id": "E-001",
319
- "target_id": "completion_time",
319
+ "target_id": "policy_effectiveness",
320
320
  "relation": "MEASURES",
321
321
  "weight": 1.0,
322
322
  "metadata": {}
@@ -351,7 +351,7 @@
351
351
  },
352
352
  {
353
353
  "source_id": "E-002",
354
- "target_id": "completion_time",
354
+ "target_id": "policy_effectiveness",
355
355
  "relation": "MEASURES",
356
356
  "weight": 1.0,
357
357
  "metadata": {}
@@ -386,7 +386,7 @@
386
386
  },
387
387
  {
388
388
  "source_id": "E-003",
389
- "target_id": "accuracy",
389
+ "target_id": "implementation_risk",
390
390
  "relation": "MEASURES",
391
391
  "weight": 1.0,
392
392
  "metadata": {}
@@ -421,7 +421,7 @@
421
421
  },
422
422
  {
423
423
  "source_id": "E-004",
424
- "target_id": "accuracy",
424
+ "target_id": "implementation_risk",
425
425
  "relation": "MEASURES",
426
426
  "weight": 1.0,
427
427
  "metadata": {}
@@ -0,0 +1,78 @@
1
+ {
2
+ "decision_question": "Should an enterprise customer-support team introduce a generative AI assistant?",
3
+ "target_population": "Enterprise customer-support staff, stratified by tenure and baseline skill.",
4
+ "target_context": "Human-supervised support using an approved knowledge base.",
5
+ "recommended_action": "pilot",
6
+ "confidence": "Moderate",
7
+ "confidence_score": 0.578,
8
+ "confidence_policy_version": "2026-08-12.v3",
9
+ "raw_model_confidence": "Moderate",
10
+ "raw_model_confidence_breakdown": {},
11
+ "independent_studies": 3,
12
+ "independent_samples": 3,
13
+ "supported_claims": [
14
+ "AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome. — E-001",
15
+ "ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support. — E-002",
16
+ "AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support. — E-003",
17
+ "Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits. — E-004"
18
+ ],
19
+ "uncertain_claims": [
20
+ "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set [no direct evidence]."
21
+ ],
22
+ "decision_rationale": "A supervised pilot is warranted because one direct field study supports efficiency gains but also reveals heterogeneous quality effects. Indirect experiments identify task boundaries, while local safety and net value remain unknown.",
23
+ "strongest_support": "Supervised AI assistance improves handling speed and answer consistency in customer-support work, with quality maintained.",
24
+ "key_uncertainty": "Evidence comes from adjacent writing and advisory settings rather than the support floor, so transfer to live customer conversations is unproven.",
25
+ "main_risk": "Unsupervised or knowledge-base-free use can produce confident wrong answers to customers, and over-reliance erodes agent skill over time.",
26
+ "next_action": "Run a supervised pilot on approved knowledge bases with human review on every reply, and track escalation and correction rates.",
27
+ "methodology_summary": "One staggered-rollout quasi-experiment and two randomized experiments. Only the support study is direct; no pooled standardized effect or model benchmark was computed.",
28
+ "what_can_be_claimed": [
29
+ "Under human supervision on an approved knowledge base, AI assistance can shorten handling time while quality is monitored.",
30
+ "Benefits are not uniform across staff: the most experienced agents need their own quality monitoring."
31
+ ],
32
+ "what_cannot_be_claimed": [
33
+ "Universal gains, autonomous deployment safety, privacy protection, reduced staffing requirements or educational learning gains."
34
+ ],
35
+ "exceeds_evidence_boundary": [
36
+ "Claiming universal gains or that autonomous deployment is safe exceeds the boundary: no included study measures privacy incidents, local net cost or subgroup service quality."
37
+ ],
38
+ "missing_evidence": [
39
+ "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set [no direct evidence]."
40
+ ],
41
+ "applicability": {
42
+ "required_conditions": [
43
+ "Start with low-risk, knowledge-base-covered queues; agents inspect and edit every proposed reply.",
44
+ "Stratify by tenure and baseline skill; retain expert discretion and measure review workload.",
45
+ "Minimize and redact customer data, restrict access and retention, and review supplier data-use terms before a pilot.",
46
+ "Escalate unfamiliar, ambiguous, sensitive or high-stakes requests to qualified humans; no autonomous commitments."
47
+ ]
48
+ },
49
+ "extensions": {
50
+ "data_origin": "manual_curated",
51
+ "benchmark_eligible": false,
52
+ "note": "Confidence and the decision bound come from the deterministic policy in engine/decision_policy.py, enforced by the Pre-Verdict Gate; this record is a curated evidence selection, not a systematic review or a model run.",
53
+ "knowledge_gaps": [
54
+ {
55
+ "gap_id": "G-001",
56
+ "evidence_ids": [
57
+ "E-001",
58
+ "E-002",
59
+ "E-003",
60
+ "E-004"
61
+ ],
62
+ "summary": "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set [no direct evidence]."
63
+ }
64
+ ]
65
+ },
66
+ "confidence_breakdown": {
67
+ "score": 0.578,
68
+ "evidence_quality": 0.8,
69
+ "consistency": 0.0,
70
+ "directness": 0.75,
71
+ "evidence_count": 4,
72
+ "independent_studies": 3,
73
+ "independent_samples": 3,
74
+ "count_term": 0.75,
75
+ "conflict_penalty": 0.0,
76
+ "unsupported_penalty": 0.0
77
+ }
78
+ }
@@ -0,0 +1,101 @@
1
+ {
2
+ "gate_version": "2026-08-13.v1",
3
+ "checked_at": "2026-09-13T03:17:13.887129+00:00",
4
+ "workspace": "examples/workplace-ai-assistant",
5
+ "items": {
6
+ "research_frame_valid": {
7
+ "title": "Research Frame valid",
8
+ "status": "pass",
9
+ "detail": "frame.json valid (question=Should an enterprise customer-support team introduce a generative AI assistant?)",
10
+ "critical": true,
11
+ "blocks_high": true
12
+ },
13
+ "sources_valid": {
14
+ "title": "Sources valid",
15
+ "status": "pass",
16
+ "detail": "3 source record(s) schema-valid",
17
+ "critical": true,
18
+ "blocks_high": true
19
+ },
20
+ "evidence_schema_valid": {
21
+ "title": "Evidence Schema valid",
22
+ "status": "pass",
23
+ "detail": "4 evidence record(s) schema-valid",
24
+ "critical": true,
25
+ "blocks_high": true
26
+ },
27
+ "source_dedupe": {
28
+ "title": "Source dedupe",
29
+ "status": "pass",
30
+ "detail": "3 unique source(s), no duplicates",
31
+ "critical": true,
32
+ "blocks_high": true
33
+ },
34
+ "counter_evidence_search": {
35
+ "title": "Counter-evidence search",
36
+ "status": "pass",
37
+ "detail": "search_performed=true; 9/9 checks run; findings=8; contradictory_evidence_found=True",
38
+ "critical": true,
39
+ "blocks_high": true
40
+ },
41
+ "methodology_audit": {
42
+ "title": "Methodology audit",
43
+ "status": "pass",
44
+ "detail": "methodology verdict=CONCERN; task/learning separated",
45
+ "critical": true,
46
+ "blocks_high": true
47
+ },
48
+ "claim_evidence_audit": {
49
+ "title": "Claim-Evidence Audit",
50
+ "status": "pass",
51
+ "detail": "all verdict claims bind to existing evidence with consistent categories",
52
+ "critical": true,
53
+ "blocks_high": false
54
+ },
55
+ "outcome_mapping": {
56
+ "title": "Outcome mapping",
57
+ "status": "warn",
58
+ "detail": "outcome keys known; frame-declared outcomes without evidence: cost_effectiveness, equity, feasibility",
59
+ "critical": false,
60
+ "blocks_high": false
61
+ },
62
+ "scope_calibration": {
63
+ "title": "Scope calibration",
64
+ "status": "pass",
65
+ "detail": "claims bounded: can=2, cannot=1, exceeds_boundary=1",
66
+ "critical": false,
67
+ "blocks_high": true
68
+ },
69
+ "independent_study_count": {
70
+ "title": "Independent study-sample count",
71
+ "status": "pass",
72
+ "detail": "independent studies=3, samples=3",
73
+ "critical": true,
74
+ "blocks_high": true
75
+ },
76
+ "deterministic_confidence": {
77
+ "title": "Deterministic confidence",
78
+ "status": "pass",
79
+ "detail": "deterministic confidence=Moderate (score=0.578, policy=2026-08-12.v3)",
80
+ "critical": true,
81
+ "blocks_high": true
82
+ },
83
+ "decision_action_consistency": {
84
+ "title": "Decision action consistency",
85
+ "status": "pass",
86
+ "detail": "action=pilot is within the conservative bound",
87
+ "critical": true,
88
+ "blocks_high": false
89
+ }
90
+ },
91
+ "passed": true,
92
+ "critical_failures": [],
93
+ "high_confidence_allowed": true,
94
+ "max_confidence": "High",
95
+ "enforcement": {
96
+ "rule": "confidence capped at {max}; High confidence requires a fully passing gate and >= 2 independent studies",
97
+ "max_confidence": "High",
98
+ "requires_action_change": false,
99
+ "action_override": null
100
+ }
101
+ }
@@ -1,55 +1,224 @@
1
1
  {
2
- "title": "Should an enterprise customer-support team introduce a generative AI assistant?",
3
- "theme": "claude",
4
- "mode": "demo",
5
- "sections": [
6
- {
7
- "id": "decision",
8
- "title": "Decision and boundaries",
9
- "component": "DecisionCard",
10
- "data_ref": "decision"
2
+ "generated_by": "build_report.py",
3
+ "source": "result.json + result.zh.json",
4
+ "question": "Should an enterprise customer-support team introduce a generative AI assistant?",
5
+ "theme_selected": "claude",
6
+ "theme_display": "Claude Research [Light]",
7
+ "theme_available": [
8
+ "claude",
9
+ "academic",
10
+ "datalab",
11
+ "datalab-dark",
12
+ "presentation"
13
+ ],
14
+ "theme_selection": "generation_time",
15
+ "lang_default": "zh",
16
+ "lang_switchable": [
17
+ "zh",
18
+ "en"
19
+ ],
20
+ "report_pages": [
21
+ "visual_brief",
22
+ "full_report"
23
+ ],
24
+ "full_report_outline": {
25
+ "chapter_count": 6,
26
+ "source": "safe_fallback",
27
+ "chapters": [
28
+ {
29
+ "key": "decision",
30
+ "title_zh": "结论、裁决与研究边界",
31
+ "title_en": "Decision, Adjudication & Research Boundary",
32
+ "modules": [
33
+ "decision",
34
+ "scope"
35
+ ]
36
+ },
37
+ {
38
+ "key": "evidence",
39
+ "title_zh": "关键证据与结果分离",
40
+ "title_en": "Key Evidence & Outcome Separation",
41
+ "modules": [
42
+ "retrieval",
43
+ "outcomes",
44
+ "evidence"
45
+ ]
46
+ },
47
+ {
48
+ "key": "quality",
49
+ "title_zh": "证据可信度、反证与方法审计",
50
+ "title_en": "Evidence Quality, Counterevidence & Method Audit",
51
+ "modules": [
52
+ "quality",
53
+ "conflicts",
54
+ "trace"
55
+ ]
56
+ },
57
+ {
58
+ "key": "action",
59
+ "title_zh": "适用范围与教学行动",
60
+ "title_en": "Applicability & Teaching Action",
61
+ "modules": [
62
+ "applicability",
63
+ "intervention"
64
+ ]
65
+ },
66
+ {
67
+ "key": "evaluation",
68
+ "title_zh": "试点设计、评估与停止条件",
69
+ "title_en": "Pilot, Evaluation & Stop Conditions",
70
+ "modules": [
71
+ "evaluation"
72
+ ]
73
+ },
74
+ {
75
+ "key": "sources",
76
+ "title_zh": "来源、溯源与附录",
77
+ "title_en": "Sources, Traceability & Appendix",
78
+ "modules": [
79
+ "sources"
80
+ ]
81
+ }
82
+ ]
83
+ },
84
+ "visualization_decisions": {
85
+ "outcome_separation": {
86
+ "render": true,
87
+ "reason": "multiple outcome constructs present"
11
88
  },
12
- {
13
- "id": "evidence",
14
- "title": "Direct and indirect evidence",
15
- "component": "EvidenceMatrix",
16
- "data_ref": "evidence"
89
+ "outcome_evidence_balance": {
90
+ "render": false,
91
+ "reason": "suppressed: sparse effect counts (total=4, active=2, nonzero_cells=2, max_cell=2)"
92
+ },
93
+ "claim_trace": {
94
+ "render": true,
95
+ "reason": "claim-evidence-source relationships present"
17
96
  },
97
+ "benchmark": {
98
+ "render": false,
99
+ "reason": "suppressed: absent, simulated, or fewer than two baselines"
100
+ }
101
+ },
102
+ "lieflat_gallery": {
103
+ "layout_source": "deterministic_fallback",
104
+ "selected": [
105
+ {
106
+ "chart_id": "lieflat-bubble-almanac.svg",
107
+ "type": "bubble_almanac",
108
+ "catalog_ref": "L9 Bubble Almanac",
109
+ "source": "evidence.year_x_dimension"
110
+ },
111
+ {
112
+ "chart_id": "lieflat-matrix-heat.svg",
113
+ "type": "matrix_heat",
114
+ "catalog_ref": "L16 Matrix Heat",
115
+ "source": "evidence.year_x_outcome_counts"
116
+ },
117
+ {
118
+ "chart_id": "lieflat-hundred-field.svg",
119
+ "type": "hundred_field",
120
+ "catalog_ref": "L14 Hundred Field",
121
+ "source": "evidence.study_type_composition"
122
+ },
123
+ {
124
+ "chart_id": "lieflat-tick-gauge.svg",
125
+ "type": "tick_gauge",
126
+ "catalog_ref": "F11 Tick Gauge",
127
+ "source": "decision.confidence_score"
128
+ },
129
+ {
130
+ "chart_id": "lieflat-ballot-tally.svg",
131
+ "type": "ballot_tally",
132
+ "catalog_ref": "L15 Ballot Tally",
133
+ "source": "methodology.flag_rates"
134
+ }
135
+ ],
136
+ "suppressed": [],
137
+ "rejected": [],
138
+ "warnings": [
139
+ "visual_layout missing or fully invalid — data-driven fallback selected bubble_almanac, matrix_heat, hundred_field, tick_gauge, ballot_tally"
140
+ ]
141
+ },
142
+ "charts": [
18
143
  {
19
- "id": "methodology_reviews",
20
- "title": "Methodological limits",
21
- "component": "MethodologyPanel",
22
- "data_ref": "methodology_reviews"
144
+ "chart_id": "outcome-evidence-overview",
145
+ "purpose": "interactive_analysis",
146
+ "engine": "echarts",
147
+ "data_ref": null,
148
+ "title": "Outcome Evidence Overview",
149
+ "integrity": {
150
+ "numbers_match_result": "NOT_CHECKED",
151
+ "no_axis_distortion": "NOT_CHECKED",
152
+ "no_false_precision": "NOT_CHECKED",
153
+ "colorblind_safe": "NOT_CHECKED"
154
+ }
23
155
  },
24
156
  {
25
- "id": "applicability",
26
- "title": "Deployment conditions",
27
- "component": "ApplicabilityCard",
28
- "data_ref": "applicability"
157
+ "chart_id": "claim-evidence-trace",
158
+ "purpose": "interactive_analysis",
159
+ "engine": "echarts",
160
+ "data_ref": null,
161
+ "title": "Claim-Evidence Trace",
162
+ "integrity": {
163
+ "numbers_match_result": "NOT_CHECKED",
164
+ "no_axis_distortion": "NOT_CHECKED",
165
+ "no_false_precision": "NOT_CHECKED",
166
+ "colorblind_safe": "NOT_CHECKED"
167
+ }
168
+ }
169
+ ],
170
+ "infographics": [
171
+ {
172
+ "chart_id": "workflow",
173
+ "purpose": "process_or_story",
174
+ "engine": "antv_infographic",
175
+ "title": ""
29
176
  },
30
177
  {
31
- "id": "intervention",
32
- "title": "Proposed pilot",
33
- "component": "InterventionTimeline",
34
- "data_ref": "intervention"
178
+ "chart_id": "tribunal",
179
+ "purpose": "process_or_story",
180
+ "engine": "antv_infographic",
181
+ "title": ""
35
182
  },
36
183
  {
37
- "id": "evaluation",
38
- "title": "Evaluation and stopping",
39
- "component": "EvaluationFlow",
40
- "data_ref": "evaluation"
184
+ "chart_id": "intervention",
185
+ "purpose": "process_or_story",
186
+ "engine": "antv_infographic",
187
+ "title": ""
41
188
  },
42
189
  {
43
- "id": "sources",
44
- "title": "Verified primary sources",
45
- "component": "SourceList",
46
- "data_ref": "sources"
190
+ "chart_id": "evaluation",
191
+ "purpose": "process_or_story",
192
+ "engine": "antv_infographic",
193
+ "title": ""
194
+ }
195
+ ],
196
+ "academic_figures": [
197
+ {
198
+ "chart_id": "outcome-comparison",
199
+ "purpose": "statistical_publication",
200
+ "engine": "academic_figure",
201
+ "caption": "Fig. 1. Counts of positive / negative / null effects per outcome type (based on effect_direction; publication figure, theme-independent). Source: EduEvidence result.json."
47
202
  }
48
203
  ],
49
- "extensions": {
50
- "data_origin": "manual_curated",
51
- "domain": "policy",
52
- "benchmark_eligible": false,
53
- "render_status": "not_generated"
204
+ "integrity_gate": {
205
+ "status": "PASS",
206
+ "contract_valid": "PASS",
207
+ "claims_bound": "PASS",
208
+ "evidence_bound": 4,
209
+ "sources_resolved": 3,
210
+ "numbers_match_result": "PASS",
211
+ "bilingual_structure_match": "PASS",
212
+ "language_match": "PASS",
213
+ "no_false_precision": "PASS",
214
+ "lieflat_data_bound": "PASS",
215
+ "no_axis_distortion": "NOT_CHECKED",
216
+ "colorblind_safe": "NOT_CHECKED",
217
+ "langs": [
218
+ "zh",
219
+ "en"
220
+ ],
221
+ "generated_by": "build_report.py",
222
+ "source": "result.json + result.zh.json"
54
223
  }
55
- }
224
+ }