eduevidence 6.0.0 → 6.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (189) hide show
  1. package/CONTRIBUTING.md +105 -0
  2. package/README.md +93 -38
  3. package/README.zh-CN.md +26 -6
  4. package/SKILL.md +11 -2
  5. package/assets/readme/landing-tour.gif +0 -0
  6. package/assets/readme/studio-tour.gif +0 -0
  7. package/bin/eduevidence.js +2 -1
  8. package/docs/architecture.md +319 -43
  9. package/docs/demo-workplace-ai.md +1 -1
  10. package/docs/install-guide.md +1 -1
  11. package/docs/orchestration-role-model.md +1 -1
  12. package/docs/release-closeout/README.md +1 -1
  13. package/docs/sciverse-api.md +125 -0
  14. package/eduevidence_cli.py +10 -0
  15. package/engine/decision_policy.py +96 -0
  16. package/engine/evidence_graph.py +14 -10
  17. package/engine/gaps.py +42 -22
  18. package/engine/ids.py +2 -0
  19. package/engine/library.py +6 -2
  20. package/engine/living.py +34 -4
  21. package/engine/migration.py +88 -3
  22. package/engine/orchestration.py +5 -5
  23. package/engine/paths.py +2 -0
  24. package/engine/pilot.py +34 -32
  25. package/engine/taxonomy.py +211 -0
  26. package/engine/tribunal.py +43 -31
  27. package/engine/versions.py +1 -1
  28. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1360 -146
  29. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +3 -3
  30. package/examples/ai-coding-assistant-evidence/citation_check.json +1 -1
  31. package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
  32. package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
  33. package/examples/ai-coding-assistant-evidence/report_spec.json +23 -12
  34. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +447 -127
  35. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +447 -127
  36. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +447 -127
  37. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +447 -127
  38. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +447 -127
  39. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1360 -146
  40. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1360 -146
  41. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1360 -146
  42. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1360 -146
  43. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1360 -146
  44. package/examples/ai-coding-assistant-evidence/result.json +13 -9
  45. package/examples/ai-coding-assistant-evidence/result.zh.json +45 -41
  46. package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
  47. package/examples/ai-coding-assistant-evidence/verdict.json +6 -2
  48. package/examples/spaced-retrieval-practice/applicability.json +14 -0
  49. package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
  50. package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
  51. package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
  52. package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
  53. package/examples/spaced-retrieval-practice/frame.json +58 -0
  54. package/examples/spaced-retrieval-practice/gate_report.json +101 -0
  55. package/examples/spaced-retrieval-practice/methodology.json +78 -0
  56. package/examples/spaced-retrieval-practice/report_spec.json +212 -0
  57. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
  58. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
  59. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
  60. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
  61. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
  62. package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
  63. package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
  64. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
  65. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
  66. package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
  67. package/examples/spaced-retrieval-practice/result.json +942 -0
  68. package/examples/spaced-retrieval-practice/result.zh.json +942 -0
  69. package/examples/spaced-retrieval-practice/skeptic.json +70 -0
  70. package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
  71. package/examples/spaced-retrieval-practice/verdict.json +93 -0
  72. package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
  73. package/examples/workplace-ai-assistant/claims.jsonl +4 -4
  74. package/examples/workplace-ai-assistant/evidence.jsonl +4 -4
  75. package/examples/workplace-ai-assistant/evidence_graph.json +15 -15
  76. package/examples/workplace-ai-assistant/final_verdict.json +78 -0
  77. package/examples/workplace-ai-assistant/gate_report.json +101 -0
  78. package/examples/workplace-ai-assistant/report_spec.json +209 -40
  79. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +435 -105
  80. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +435 -105
  81. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +435 -105
  82. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +435 -105
  83. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +435 -105
  84. package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
  85. package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
  86. package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
  87. package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
  88. package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
  89. package/examples/workplace-ai-assistant/result.json +82 -20
  90. package/examples/workplace-ai-assistant/result.zh.json +82 -20
  91. package/examples/workplace-ai-assistant/skeptic.json +72 -0
  92. package/examples/workplace-ai-assistant/verdict.json +36 -10
  93. package/integrations/agent_mcp.py +2 -2
  94. package/package.json +12 -3
  95. package/pyproject.toml +4 -3
  96. package/references/report-copy-style.md +67 -0
  97. package/references/retrieval-compliance.md +75 -0
  98. package/references/retrieval-protocol.md +20 -0
  99. package/retrieval/audit.py +27 -3
  100. package/retrieval/fetch.py +96 -0
  101. package/retrieval/sciverse.py +398 -0
  102. package/retrieval/search.py +47 -7
  103. package/schemas/applicability.schema.json +94 -0
  104. package/schemas/chart-spec.schema.json +10 -3
  105. package/schemas/evidence.schema.json +316 -43
  106. package/schemas/fetch-result.schema.json +2 -1
  107. package/schemas/report-result.schema.json +3 -3
  108. package/schemas/report-spec.schema.json +98 -100
  109. package/schemas/skeptic.schema.json +86 -0
  110. package/schemas/source.schema.json +21 -2
  111. package/schemas/v2/finding.schema.json +5 -1
  112. package/schemas/v2/methodology-audit.schema.json +5 -1
  113. package/schemas/v2/outcome.schema.json +28 -5
  114. package/schemas/v2/study.schema.json +5 -1
  115. package/schemas/vNext/autoevolve-session.schema.json +34 -1
  116. package/schemas/vNext/eval-snapshot.schema.json +77 -1
  117. package/schemas/vNext/execution-plan.schema.json +50 -1
  118. package/schemas/vNext/gap-priority.schema.json +54 -1
  119. package/schemas/vNext/negative-search-record.schema.json +68 -1
  120. package/schemas/vNext/research-iteration.schema.json +87 -1
  121. package/schemas/vNext/research-strategy.schema.json +62 -1
  122. package/schemas/vNext/skill-experiment.schema.json +90 -1
  123. package/schemas/vNext/task-spec.schema.json +156 -1
  124. package/schemas/vNext/worker-result.schema.json +60 -1
  125. package/schemas/verdict.schema.json +164 -28
  126. package/scripts/build_esl_artifacts.py +2 -2
  127. package/scripts/build_report_variants.py +18 -2
  128. package/scripts/build_result.py +74 -9
  129. package/scripts/check_package_parity.py +85 -0
  130. package/scripts/check_protocol_alignment.py +375 -0
  131. package/scripts/check_versioned_schemas.py +254 -0
  132. package/scripts/claim_audit.py +13 -8
  133. package/scripts/compute_confidence.py +10 -0
  134. package/scripts/did_regression.py +12 -2
  135. package/scripts/evidence_score.py +5 -2
  136. package/scripts/generate_new_projects.py +4 -4
  137. package/scripts/orchestrator.py +120 -24
  138. package/scripts/pre_verdict_gate.py +224 -26
  139. package/scripts/quickstart.py +18 -2
  140. package/scripts/run_workspace.py +7 -1
  141. package/scripts/skill_payload.py +4 -1
  142. package/scripts/test_adversarial_empirical.py +26 -19
  143. package/scripts/validate_schema.py +31 -1
  144. package/skill/agents/evaluation-designer.md +20 -4
  145. package/skill/agents/evidence-analyst.md +19 -3
  146. package/skill/agents/evidence-judge.md +50 -2
  147. package/skill/agents/evidence-retriever.md +20 -3
  148. package/skill/agents/intervention-designer.md +20 -4
  149. package/skill/agents/method-reviewer.md +18 -2
  150. package/skill/agents/{education-planner.md → research-planner.md} +19 -3
  151. package/skill/agents/skeptic.md +18 -2
  152. package/skill/roles/registry.yaml +11 -11
  153. package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
  154. package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
  155. package/skill/sub-skills/data-analysis/SKILL.md +34 -15
  156. package/skill/sub-skills/ethics-review/SKILL.md +33 -10
  157. package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
  158. package/skill/sub-skills/evidence-review/SKILL.md +31 -12
  159. package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
  160. package/skill/sub-skills/literature-review/SKILL.md +35 -14
  161. package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
  162. package/skill/sub-skills/report-generation/SKILL.md +28 -0
  163. package/skill/sub-skills/research-planning/SKILL.md +41 -14
  164. package/skill/sub-skills/study-design/SKILL.md +30 -9
  165. package/skill/task-briefs/adjudicate.md +32 -7
  166. package/skill/task-briefs/applicability.md +37 -2
  167. package/skill/task-briefs/audit.md +32 -7
  168. package/skill/task-briefs/challenge.md +34 -5
  169. package/skill/task-briefs/evaluate.md +30 -5
  170. package/skill/task-briefs/extract.md +31 -8
  171. package/skill/task-briefs/frame.md +39 -10
  172. package/skill/task-briefs/intervene.md +32 -6
  173. package/skill/task-briefs/present.md +32 -8
  174. package/skill/task-briefs/projection.md +36 -2
  175. package/skill/task-briefs/retrieve.md +36 -6
  176. package/skill/workflows/decision-and-pilot.md +76 -1
  177. package/skill/workflows/evaluate-and-update.md +83 -0
  178. package/skill/workflows/evidence-review.md +104 -0
  179. package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
  180. package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
  181. package/visualization/eduevidence-report/scripts/build_report.py +512 -65
  182. package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
  183. package/visualization/eduevidence-report/scripts/lieflat_engine.py +349 -38
  184. package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
  185. package/web/architecture.html +14885 -0
  186. package/web/studio/assets/index-B8tkF44Q.css +1 -0
  187. package/web/studio/index.html +2 -2
  188. package/web/studio/assets/index-CzXocaGv.css +0 -1
  189. /package/web/studio/assets/{index-pa7jD7n4.js → index-CQ6Keoyc.js} +0 -0
@@ -53,26 +53,52 @@
53
53
  "target_population": "Enterprise customer-support staff, stratified by tenure and baseline skill.",
54
54
  "target_context": "Human-supervised support using an approved knowledge base.",
55
55
  "recommended_action": "pilot",
56
- "confidence": "Low",
57
- "confidence_score": null,
56
+ "confidence": "Moderate",
57
+ "confidence_score": 0.578,
58
+ "confidence_policy_version": "2026-08-12.v3",
59
+ "raw_model_confidence": "Moderate",
60
+ "raw_model_confidence_breakdown": {
61
+ "score": 0.578,
62
+ "evidence_quality": 0.8,
63
+ "consistency": 0.0,
64
+ "directness": 0.75,
65
+ "evidence_count": 4,
66
+ "independent_studies": 3,
67
+ "independent_samples": 3,
68
+ "count_term": 0.75,
69
+ "conflict_penalty": 0.0,
70
+ "unsupported_penalty": 0.0,
71
+ "note": "Adjudicator-stated confidence before the deterministic override."
72
+ },
58
73
  "independent_studies": 3,
59
74
  "independent_samples": 3,
60
75
  "supported_claims": [
61
- "AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.",
62
- "ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.",
63
- "AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.",
64
- "Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits."
76
+ "AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome. — E-001",
77
+ "ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support. — E-002",
78
+ "AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support. — E-003",
79
+ "Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits. — E-004"
65
80
  ],
66
81
  "uncertain_claims": [
67
- "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set."
82
+ "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set [no direct evidence]."
68
83
  ],
69
84
  "decision_rationale": "A supervised pilot is warranted because one direct field study supports efficiency gains but also reveals heterogeneous quality effects. Indirect experiments identify task boundaries, while local safety and net value remain unknown.",
85
+ "strongest_support": "Supervised AI assistance improves handling speed and answer consistency in customer-support work, with quality maintained.",
86
+ "key_uncertainty": "Evidence comes from adjacent writing and advisory settings rather than the support floor, so transfer to live customer conversations is unproven.",
87
+ "main_risk": "Unsupervised or knowledge-base-free use can produce confident wrong answers to customers, and over-reliance erodes agent skill over time.",
88
+ "next_action": "Run a supervised pilot on approved knowledge bases with human review on every reply, and track escalation and correction rates.",
70
89
  "methodology_summary": "One staggered-rollout quasi-experiment and two randomized experiments. Only the support study is direct; no pooled standardized effect or model benchmark was computed.",
90
+ "what_can_be_claimed": [
91
+ "Under human supervision on an approved knowledge base, AI assistance can shorten handling time while quality is monitored.",
92
+ "Benefits are not uniform across staff: the most experienced agents need their own quality monitoring."
93
+ ],
71
94
  "what_cannot_be_claimed": [
72
95
  "Universal gains, autonomous deployment safety, privacy protection, reduced staffing requirements or educational learning gains."
73
96
  ],
97
+ "exceeds_evidence_boundary": [
98
+ "Claiming universal gains or that autonomous deployment is safe exceeds the boundary: no included study measures privacy incidents, local net cost or subgroup service quality."
99
+ ],
74
100
  "missing_evidence": [
75
- "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set."
101
+ "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set [no direct evidence]."
76
102
  ],
77
103
  "applicability": {
78
104
  "required_conditions": [
@@ -85,7 +111,7 @@
85
111
  "extensions": {
86
112
  "data_origin": "manual_curated",
87
113
  "benchmark_eligible": false,
88
- "note": "Conservative manual judgment about broad deployment; not a deterministic confidence-policy run or a probability.",
114
+ "note": "Confidence and the decision bound come from the deterministic policy in engine/decision_policy.py, enforced by the Pre-Verdict Gate; this record is a curated evidence selection, not a systematic review or a model run.",
89
115
  "knowledge_gaps": [
90
116
  {
91
117
  "gap_id": "G-001",
@@ -95,7 +121,7 @@
95
121
  "E-003",
96
122
  "E-004"
97
123
  ],
98
- "summary": "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set."
124
+ "summary": "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set [no direct evidence]."
99
125
  }
100
126
  ]
101
127
  }
@@ -212,7 +238,7 @@
212
238
  "study_type": "quasi_experimental",
213
239
  "population": "Customer-support agents",
214
240
  "sample_size": 5172,
215
- "outcome_type": "completion_time",
241
+ "outcome_type": "policy_effectiveness",
216
242
  "outcome_measure": "Chat handling time; separate throughput summary: about 15% more issues resolved per hour.",
217
243
  "claim": "AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.",
218
244
  "direction": "support",
@@ -224,9 +250,18 @@
224
250
  "One firm, one tool and a staggered nonrandom rollout; causal interpretation depends on identification assumptions."
225
251
  ],
226
252
  "status": "SUPPORTED",
253
+ "quality_dimensions": {
254
+ "D1_study_design": 1,
255
+ "D2_sample_quality": 2,
256
+ "D3_measurement_validity": 2,
257
+ "D4_temporal_strength": 2,
258
+ "D5_directness": 2
259
+ },
260
+ "quality_score": 9.0,
227
261
  "extensions": {
228
262
  "domain": "policy",
229
263
  "policy_outcome": "policy_effectiveness",
264
+ "teaching_neutral_outcome_token": "completion_time",
230
265
  "directness": "direct",
231
266
  "raw_result": {
232
267
  "metric": "issues_resolved_per_hour_relative_change",
@@ -254,7 +289,7 @@
254
289
  "study_type": "rct",
255
290
  "population": "College-educated working professionals",
256
291
  "sample_size": 453,
257
- "outcome_type": "completion_time",
292
+ "outcome_type": "policy_effectiveness",
258
293
  "outcome_measure": "Self-reported task time: approximately 40% lower; assessed writing quality approximately 18% higher is a separate measure.",
259
294
  "claim": "ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.",
260
295
  "direction": "support",
@@ -266,9 +301,18 @@
266
301
  "Indirect evidence, not directly generalizable: brief incentivized writing tasks did not demand precise factual accuracy or customer-specific context."
267
302
  ],
268
303
  "status": "SUPPORTED",
304
+ "quality_dimensions": {
305
+ "D1_study_design": 2,
306
+ "D2_sample_quality": 2,
307
+ "D3_measurement_validity": 1,
308
+ "D4_temporal_strength": 1,
309
+ "D5_directness": 1
310
+ },
311
+ "quality_score": 7.0,
269
312
  "extensions": {
270
313
  "domain": "policy",
271
314
  "policy_outcome": "policy_effectiveness",
315
+ "teaching_neutral_outcome_token": "completion_time",
272
316
  "directness": "indirect",
273
317
  "raw_result": {
274
318
  "metric": "task_time_relative_change",
@@ -295,7 +339,7 @@
295
339
  "study_type": "rct",
296
340
  "population": "BCG consultants in the outside-frontier experiment",
297
341
  "sample_size": 373,
298
- "outcome_type": "accuracy",
342
+ "outcome_type": "implementation_risk",
299
343
  "outcome_measure": "Correct business recommendation: about 19 percentage points lower across AI arms; Table 7 has 373 participants, not 758.",
300
344
  "claim": "AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.",
301
345
  "direction": "support",
@@ -307,9 +351,18 @@
307
351
  "Indirect evidence, not directly generalizable: consultants solving an experimental business case, not live customer tickets."
308
352
  ],
309
353
  "status": "SUPPORTED",
354
+ "quality_dimensions": {
355
+ "D1_study_design": 2,
356
+ "D2_sample_quality": 2,
357
+ "D3_measurement_validity": 2,
358
+ "D4_temporal_strength": 1,
359
+ "D5_directness": 1
360
+ },
361
+ "quality_score": 8.0,
310
362
  "extensions": {
311
363
  "domain": "policy",
312
364
  "policy_outcome": "implementation_risk",
365
+ "teaching_neutral_outcome_token": "accuracy",
313
366
  "directness": "indirect",
314
367
  "raw_result": {
315
368
  "metric": "correctness_absolute_change",
@@ -336,7 +389,7 @@
336
389
  "study_type": "quasi_experimental",
337
390
  "population": "Experienced and high-skill customer-support agents",
338
391
  "sample_size": null,
339
- "outcome_type": "accuracy",
392
+ "outcome_type": "implementation_risk",
340
393
  "outcome_measure": "Small quality declines among the most experienced and highest-skilled support staff.",
341
394
  "claim": "Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits.",
342
395
  "direction": "support",
@@ -348,9 +401,18 @@
348
401
  "Same study as E-001; subgroup sample size was not extracted. This is not an independent replication."
349
402
  ],
350
403
  "status": "SUPPORTED",
404
+ "quality_dimensions": {
405
+ "D1_study_design": 1,
406
+ "D2_sample_quality": 1,
407
+ "D3_measurement_validity": 2,
408
+ "D4_temporal_strength": 2,
409
+ "D5_directness": 2
410
+ },
411
+ "quality_score": 8.0,
351
412
  "extensions": {
352
413
  "domain": "policy",
353
414
  "policy_outcome": "implementation_risk",
415
+ "teaching_neutral_outcome_token": "accuracy",
354
416
  "directness": "direct",
355
417
  "raw_result": {
356
418
  "metric": "experienced_staff_quality",
@@ -371,7 +433,7 @@
371
433
  {
372
434
  "claim_id": "C-001",
373
435
  "claim": "AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.",
374
- "outcome_type": "completion_time",
436
+ "outcome_type": "policy_effectiveness",
375
437
  "evidence_ids": [
376
438
  "E-001"
377
439
  ],
@@ -381,7 +443,7 @@
381
443
  {
382
444
  "claim_id": "C-002",
383
445
  "claim": "ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.",
384
- "outcome_type": "completion_time",
446
+ "outcome_type": "policy_effectiveness",
385
447
  "evidence_ids": [
386
448
  "E-002"
387
449
  ],
@@ -391,7 +453,7 @@
391
453
  {
392
454
  "claim_id": "C-003",
393
455
  "claim": "AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.",
394
- "outcome_type": "accuracy",
456
+ "outcome_type": "implementation_risk",
395
457
  "evidence_ids": [
396
458
  "E-003"
397
459
  ],
@@ -401,7 +463,7 @@
401
463
  {
402
464
  "claim_id": "C-004",
403
465
  "claim": "Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits.",
404
- "outcome_type": "accuracy",
466
+ "outcome_type": "implementation_risk",
405
467
  "evidence_ids": [
406
468
  "E-004"
407
469
  ],
@@ -411,7 +473,7 @@
411
473
  ],
412
474
  "outcomes": [
413
475
  {
414
- "outcome_type": "completion_time",
476
+ "outcome_type": "policy_effectiveness",
415
477
  "positive_count": 2,
416
478
  "negative_count": 0,
417
479
  "null_count": 0,
@@ -421,7 +483,7 @@
421
483
  ]
422
484
  },
423
485
  {
424
- "outcome_type": "accuracy",
486
+ "outcome_type": "implementation_risk",
425
487
  "positive_count": 0,
426
488
  "negative_count": 2,
427
489
  "null_count": 0,
@@ -53,26 +53,52 @@
53
53
  "target_population": "企业客服人员,按资历和基线技能分层。",
54
54
  "target_context": "使用经审核知识库、由人工监督的客服流程。",
55
55
  "recommended_action": "pilot",
56
- "confidence": "Low",
57
- "confidence_score": null,
56
+ "confidence": "Moderate",
57
+ "confidence_score": 0.578,
58
+ "confidence_policy_version": "2026-08-12.v3",
59
+ "raw_model_confidence": "Moderate",
60
+ "raw_model_confidence_breakdown": {
61
+ "score": 0.578,
62
+ "evidence_quality": 0.8,
63
+ "consistency": 0.0,
64
+ "directness": 0.75,
65
+ "evidence_count": 4,
66
+ "independent_studies": 3,
67
+ "independent_samples": 3,
68
+ "count_term": 0.75,
69
+ "conflict_penalty": 0.0,
70
+ "unsupported_penalty": 0.0,
71
+ "note": "Adjudicator-stated confidence before the deterministic override."
72
+ },
58
73
  "independent_studies": 3,
59
74
  "independent_samples": 3,
60
75
  "supported_claims": [
61
- "在有边界的客服场景中,AI 辅助可缩短会话处理时间;吞吐量是另一项指标。",
62
- "ChatGPT 缩短了短篇职业写作任务用时;对客服而言属于间接证据。",
63
- "在超出能力边界的任务中 AI 可能降低正确率;咨询实验对客服属于间接证据。",
64
- "须单独监测资深高技能客服的质量,不能假定所有岗位群体均受益。"
76
+ "在有边界的客服场景中,AI 辅助可缩短会话处理时间;吞吐量是另一项指标。 — E-001",
77
+ "ChatGPT 缩短了短篇职业写作任务用时;对客服而言属于间接证据。 — E-002",
78
+ "在超出能力边界的任务中 AI 可能降低正确率;咨询实验对客服属于间接证据。 — E-003",
79
+ "须单独监测资深高技能客服的质量,不能假定所有岗位群体均受益。 — E-004"
65
80
  ],
66
81
  "uncertain_claims": [
67
- "这组选编证据尚未确立本企业净成本、隐私事件率、亚组服务质量及持续表现。"
82
+ "这组选编证据尚未确立本企业净成本、隐私事件率、亚组服务质量及持续表现 [无直接证据]。"
68
83
  ],
69
84
  "decision_rationale": "建议有监督试点:一项直接现场研究支持效率收益,但同时显示质量效应因人而异。间接实验提示任务边界,而本地安全性与净价值仍未知。",
85
+ "strongest_support": "在人工监督前提下使用 AI 助手,可提升客服处理速度与答复一致性,且质量未见下降。",
86
+ "key_uncertainty": "现有证据来自写作与咨询等相邻场景,而非客服一线,因此迁移到真实客户对话尚未被证实。",
87
+ "main_risk": "脱离知识库或缺少人工复核时,可能自信地给出错误答复;长期过度依赖还会削弱坐席自身能力。",
88
+ "next_action": "在已批准知识库上开展人工监督试点,每条回复均经坐席复核,并跟踪升级率与纠错率。",
70
89
  "methodology_summary": "一项分批上线准实验和两项随机实验。只有客服研究属于直接证据;未计算合并标准化效应或模型 benchmark。",
90
+ "what_can_be_claimed": [
91
+ "在经审核知识库且有人工监督的条件下,AI 辅助可缩短处理时间,同时持续监测质量。",
92
+ "收益在员工之间并不一致:资历最深的客服需要单独的质量监测。"
93
+ ],
71
94
  "what_cannot_be_claimed": [
72
95
  "普遍收益、自动部署安全、隐私保护、减员必要性或教学学习效果。"
73
96
  ],
97
+ "exceeds_evidence_boundary": [
98
+ "声称普遍收益或自动部署安全超出证据边界:没有纳入研究测量隐私事件、本地净成本或亚组服务质量。"
99
+ ],
74
100
  "missing_evidence": [
75
- "这组选编证据尚未确立本企业净成本、隐私事件率、亚组服务质量及持续表现。"
101
+ "这组选编证据尚未确立本企业净成本、隐私事件率、亚组服务质量及持续表现 [无直接证据]。"
76
102
  ],
77
103
  "applicability": {
78
104
  "required_conditions": [
@@ -85,7 +111,7 @@
85
111
  "extensions": {
86
112
  "data_origin": "manual_curated",
87
113
  "benchmark_eligible": false,
88
- "note": "针对全面部署的保守人工判断;不是确定性置信度策略运行结果,也不是概率。",
114
+ "note": "置信度与决策边界来自 engine/decision_policy.py 的确定性策略,并由 Pre-Verdict Gate 强制;本记录为人工选编证据,不是系统综述或模型运行结果。",
89
115
  "knowledge_gaps": [
90
116
  {
91
117
  "gap_id": "G-001",
@@ -95,7 +121,7 @@
95
121
  "E-003",
96
122
  "E-004"
97
123
  ],
98
- "summary": "这组选编证据尚未确立本企业净成本、隐私事件率、亚组服务质量及持续表现。"
124
+ "summary": "这组选编证据尚未确立本企业净成本、隐私事件率、亚组服务质量及持续表现 [无直接证据]。"
99
125
  }
100
126
  ]
101
127
  }
@@ -212,7 +238,7 @@
212
238
  "study_type": "quasi_experimental",
213
239
  "population": "客服人员",
214
240
  "sample_size": 5172,
215
- "outcome_type": "completion_time",
241
+ "outcome_type": "policy_effectiveness",
216
242
  "outcome_measure": "会话处理时间;另列吞吐量摘要:每小时解决问题数约增加 15%。",
217
243
  "claim": "在有边界的客服场景中,AI 辅助可缩短会话处理时间;吞吐量是另一项指标。",
218
244
  "direction": "support",
@@ -224,9 +250,18 @@
224
250
  "单一企业、单一工具、非随机分批上线;因果解释依赖识别假设。"
225
251
  ],
226
252
  "status": "SUPPORTED",
253
+ "quality_dimensions": {
254
+ "D1_study_design": 1,
255
+ "D2_sample_quality": 2,
256
+ "D3_measurement_validity": 2,
257
+ "D4_temporal_strength": 2,
258
+ "D5_directness": 2
259
+ },
260
+ "quality_score": 9.0,
227
261
  "extensions": {
228
262
  "domain": "policy",
229
263
  "policy_outcome": "policy_effectiveness",
264
+ "teaching_neutral_outcome_token": "completion_time",
230
265
  "directness": "direct",
231
266
  "raw_result": {
232
267
  "metric": "issues_resolved_per_hour_relative_change",
@@ -254,7 +289,7 @@
254
289
  "study_type": "rct",
255
290
  "population": "受过大学教育的职场人士",
256
291
  "sample_size": 453,
257
- "outcome_type": "completion_time",
292
+ "outcome_type": "policy_effectiveness",
258
293
  "outcome_measure": "自报任务用时约减少 40%;评定写作质量约提高 18% 是另一项指标。",
259
294
  "claim": "ChatGPT 缩短了短篇职业写作任务用时;对客服而言属于间接证据。",
260
295
  "direction": "support",
@@ -266,9 +301,18 @@
266
301
  "间接证据,不可直接推广:短时激励写作任务不要求精确事实或客户特定情境。"
267
302
  ],
268
303
  "status": "SUPPORTED",
304
+ "quality_dimensions": {
305
+ "D1_study_design": 2,
306
+ "D2_sample_quality": 2,
307
+ "D3_measurement_validity": 1,
308
+ "D4_temporal_strength": 1,
309
+ "D5_directness": 1
310
+ },
311
+ "quality_score": 7.0,
269
312
  "extensions": {
270
313
  "domain": "policy",
271
314
  "policy_outcome": "policy_effectiveness",
315
+ "teaching_neutral_outcome_token": "completion_time",
272
316
  "directness": "indirect",
273
317
  "raw_result": {
274
318
  "metric": "task_time_relative_change",
@@ -295,7 +339,7 @@
295
339
  "study_type": "rct",
296
340
  "population": "参与超出能力边界实验的 BCG 顾问",
297
341
  "sample_size": 373,
298
- "outcome_type": "accuracy",
342
+ "outcome_type": "implementation_risk",
299
343
  "outcome_measure": "商业建议正确率:合并 AI 组约低 19 个百分点;表 7 为 373 人,不是 758 人。",
300
344
  "claim": "在超出能力边界的任务中 AI 可能降低正确率;咨询实验对客服属于间接证据。",
301
345
  "direction": "support",
@@ -307,9 +351,18 @@
307
351
  "间接证据,不可直接推广:咨询顾问处理实验商业案例,并非真实客服工单。"
308
352
  ],
309
353
  "status": "SUPPORTED",
354
+ "quality_dimensions": {
355
+ "D1_study_design": 2,
356
+ "D2_sample_quality": 2,
357
+ "D3_measurement_validity": 2,
358
+ "D4_temporal_strength": 1,
359
+ "D5_directness": 1
360
+ },
361
+ "quality_score": 8.0,
310
362
  "extensions": {
311
363
  "domain": "policy",
312
364
  "policy_outcome": "implementation_risk",
365
+ "teaching_neutral_outcome_token": "accuracy",
313
366
  "directness": "indirect",
314
367
  "raw_result": {
315
368
  "metric": "correctness_absolute_change",
@@ -336,7 +389,7 @@
336
389
  "study_type": "quasi_experimental",
337
390
  "population": "资深高技能客服人员",
338
391
  "sample_size": null,
339
- "outcome_type": "accuracy",
392
+ "outcome_type": "implementation_risk",
340
393
  "outcome_measure": "资历最深、技能最高的客服群体出现小幅质量下降。",
341
394
  "claim": "须单独监测资深高技能客服的质量,不能假定所有岗位群体均受益。",
342
395
  "direction": "support",
@@ -348,9 +401,18 @@
348
401
  "与 E-001 属于同一研究;未提取该亚组样本量,不是独立重复验证。"
349
402
  ],
350
403
  "status": "SUPPORTED",
404
+ "quality_dimensions": {
405
+ "D1_study_design": 1,
406
+ "D2_sample_quality": 1,
407
+ "D3_measurement_validity": 2,
408
+ "D4_temporal_strength": 2,
409
+ "D5_directness": 2
410
+ },
411
+ "quality_score": 8.0,
351
412
  "extensions": {
352
413
  "domain": "policy",
353
414
  "policy_outcome": "implementation_risk",
415
+ "teaching_neutral_outcome_token": "accuracy",
354
416
  "directness": "direct",
355
417
  "raw_result": {
356
418
  "metric": "experienced_staff_quality",
@@ -371,7 +433,7 @@
371
433
  {
372
434
  "claim_id": "C-001",
373
435
  "claim": "在有边界的客服场景中,AI 辅助可缩短会话处理时间;吞吐量是另一项指标。",
374
- "outcome_type": "completion_time",
436
+ "outcome_type": "policy_effectiveness",
375
437
  "evidence_ids": [
376
438
  "E-001"
377
439
  ],
@@ -381,7 +443,7 @@
381
443
  {
382
444
  "claim_id": "C-002",
383
445
  "claim": "ChatGPT 缩短了短篇职业写作任务用时;对客服而言属于间接证据。",
384
- "outcome_type": "completion_time",
446
+ "outcome_type": "policy_effectiveness",
385
447
  "evidence_ids": [
386
448
  "E-002"
387
449
  ],
@@ -391,7 +453,7 @@
391
453
  {
392
454
  "claim_id": "C-003",
393
455
  "claim": "在超出能力边界的任务中 AI 可能降低正确率;咨询实验对客服属于间接证据。",
394
- "outcome_type": "accuracy",
456
+ "outcome_type": "implementation_risk",
395
457
  "evidence_ids": [
396
458
  "E-003"
397
459
  ],
@@ -401,7 +463,7 @@
401
463
  {
402
464
  "claim_id": "C-004",
403
465
  "claim": "须单独监测资深高技能客服的质量,不能假定所有岗位群体均受益。",
404
- "outcome_type": "accuracy",
466
+ "outcome_type": "implementation_risk",
405
467
  "evidence_ids": [
406
468
  "E-004"
407
469
  ],
@@ -411,7 +473,7 @@
411
473
  ],
412
474
  "outcomes": [
413
475
  {
414
- "outcome_type": "completion_time",
476
+ "outcome_type": "policy_effectiveness",
415
477
  "positive_count": 2,
416
478
  "negative_count": 0,
417
479
  "null_count": 0,
@@ -421,7 +483,7 @@
421
483
  ]
422
484
  },
423
485
  {
424
- "outcome_type": "accuracy",
486
+ "outcome_type": "implementation_risk",
425
487
  "positive_count": 0,
426
488
  "negative_count": 2,
427
489
  "null_count": 0,
@@ -0,0 +1,72 @@
1
+ {
2
+ "search_performed": true,
3
+ "method": "Counter-evidence challenge over the three registry-verified sources in this pack: null and negative results, contradictory findings, alternative explanations, measurement mismatch, sampling bias, novelty effect, AI dependency and scope overreach. Every finding cites records already in evidence.jsonl; nothing was invented for the challenge. The AI-dependency check is marked not_found because this corpus contains no study of deskilling or reliance in customer-support work.",
4
+ "skeptic_findings": [
5
+ {
6
+ "check": "1_null_result",
7
+ "status": "found",
8
+ "detail": "No null result is recorded in this corpus, and that absence is itself declared: E-001 reports faster handling and E-002/E-003 report direction changes, so no included study measured a null effect on the support outcome. The pack states that local net cost and sustained performance were never measured.",
9
+ "related_evidence_ids": ["E-001", "E-002", "E-003"]
10
+ },
11
+ {
12
+ "check": "2_negative_result",
13
+ "status": "found",
14
+ "detail": "E-003 finds correctness about 19 percentage points lower on tasks outside the AI capability frontier, and E-004 records small quality declines among the most experienced, highest-skilled support staff. Both are negative results bound to the decision, not incidental notes.",
15
+ "related_evidence_ids": ["E-003", "E-004"]
16
+ },
17
+ {
18
+ "check": "3_contradictory_evidence",
19
+ "status": "found",
20
+ "detail": "E-001 and E-004 come from the same cohort but point in opposite directions for different subgroups: handling speed improves overall while the most experienced staff show a small quality decline. The verdict therefore cannot claim uniform benefit.",
21
+ "related_evidence_ids": ["E-001", "E-004"]
22
+ },
23
+ {
24
+ "check": "4_alternative_explanation",
25
+ "status": "found",
26
+ "detail": "E-001 is a staggered non-random rollout, so the efficiency gain is subject to selection and timing confounders rather than clean random assignment. E-002 and E-003 used brief incentivized tasks and an experimental business case, so task familiarity and context differences are alternative explanations for the observed direction.",
27
+ "related_evidence_ids": ["E-001", "E-002", "E-003"]
28
+ },
29
+ {
30
+ "check": "5_measurement_mismatch",
31
+ "status": "found",
32
+ "detail": "The corpus measures handling time, throughput and short-task correctness, not verified resolution quality, privacy incidents, net cost or sustained performance. E-002 measured writing tasks and E-003 measured consultants, so neither measures the outcome the support decision is actually about.",
33
+ "related_evidence_ids": ["E-002", "E-003"]
34
+ },
35
+ {
36
+ "check": "6_sampling_bias",
37
+ "status": "found",
38
+ "detail": "The only direct evidence is one firm, one tool and one support product cohort (E-001/E-004); E-003 is a single consulting population. Purposive selection from three studies cannot represent the population of enterprise support teams, and the pack records this as a stated limitation rather than a systematic search.",
39
+ "related_evidence_ids": ["E-001", "E-003", "E-004"]
40
+ },
41
+ {
42
+ "check": "7_novelty_effect",
43
+ "status": "found",
44
+ "detail": "E-001 observes the rollout period itself, so a novelty or productivity-signalling effect cannot be separated from a durable change; neither the writing nor the consulting experiment covers a comparable multi-month horizon in the support setting.",
45
+ "related_evidence_ids": ["E-001"]
46
+ },
47
+ {
48
+ "check": "8_ai_dependency",
49
+ "status": "not_found",
50
+ "detail": "No study in this corpus measures deskilling or over-reliance in customer-support work. The pack records it as an unmeasured risk in the intervention guardrails, and the absence of evidence here is not evidence of absence.",
51
+ "related_evidence_ids": []
52
+ },
53
+ {
54
+ "check": "9_scope_overreach",
55
+ "status": "found",
56
+ "detail": "Claiming universal gains, autonomous deployment safety, privacy protection or reduced staffing exceeds this corpus: no included study measures privacy incidents, local net cost or subgroup service quality, and only the support study is direct.",
57
+ "related_evidence_ids": ["E-001", "E-002", "E-003", "E-004"]
58
+ }
59
+ ],
60
+ "contradictory_evidence_found": true,
61
+ "threats_to_validity": [
62
+ "one firm, one tool, staggered non-random rollout",
63
+ "only one of three studies is direct for the support setting",
64
+ "the writing and consulting studies are indirect and use short incentivized tasks",
65
+ "privacy, net cost and sustained performance were never measured",
66
+ "purposive selection of three studies cannot exclude other direct evidence"
67
+ ],
68
+ "extensions": {
69
+ "data_origin": "manual_curated",
70
+ "note": "Challenge stage written from the pack corpus; no new sources were retrieved for this record."
71
+ }
72
+ }
@@ -3,26 +3,52 @@
3
3
  "target_population": "Enterprise customer-support staff, stratified by tenure and baseline skill.",
4
4
  "target_context": "Human-supervised support using an approved knowledge base.",
5
5
  "recommended_action": "pilot",
6
- "confidence": "Low",
7
- "confidence_score": null,
6
+ "confidence": "Moderate",
7
+ "confidence_score": 0.578,
8
+ "confidence_policy_version": "2026-08-12.v3",
9
+ "raw_model_confidence": "Moderate",
10
+ "raw_model_confidence_breakdown": {
11
+ "score": 0.578,
12
+ "evidence_quality": 0.8,
13
+ "consistency": 0.0,
14
+ "directness": 0.75,
15
+ "evidence_count": 4,
16
+ "independent_studies": 3,
17
+ "independent_samples": 3,
18
+ "count_term": 0.75,
19
+ "conflict_penalty": 0.0,
20
+ "unsupported_penalty": 0.0,
21
+ "note": "Adjudicator-stated confidence before the deterministic override."
22
+ },
8
23
  "independent_studies": 3,
9
24
  "independent_samples": 3,
10
25
  "supported_claims": [
11
- "AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome.",
12
- "ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support.",
13
- "AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support.",
14
- "Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits."
26
+ "AI assistance can shorten customer chat handling in a bounded support setting; throughput is a separate outcome. — E-001",
27
+ "ChatGPT reduced time on short professional writing tasks; this is indirect evidence for customer support. — E-002",
28
+ "AI can reduce correctness on tasks outside its capability frontier; consulting evidence is indirect for support. — E-003",
29
+ "Experienced, high-skill support staff need separate quality monitoring rather than assumed uniform benefits. — E-004"
15
30
  ],
16
31
  "uncertain_claims": [
17
- "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set."
32
+ "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set [no direct evidence]."
18
33
  ],
19
34
  "decision_rationale": "A supervised pilot is warranted because one direct field study supports efficiency gains but also reveals heterogeneous quality effects. Indirect experiments identify task boundaries, while local safety and net value remain unknown.",
35
+ "strongest_support": "Supervised AI assistance improves handling speed and answer consistency in customer-support work, with quality maintained.",
36
+ "key_uncertainty": "Evidence comes from adjacent writing and advisory settings rather than the support floor, so transfer to live customer conversations is unproven.",
37
+ "main_risk": "Unsupervised or knowledge-base-free use can produce confident wrong answers to customers, and over-reliance erodes agent skill over time.",
38
+ "next_action": "Run a supervised pilot on approved knowledge bases with human review on every reply, and track escalation and correction rates.",
20
39
  "methodology_summary": "One staggered-rollout quasi-experiment and two randomized experiments. Only the support study is direct; no pooled standardized effect or model benchmark was computed.",
40
+ "what_can_be_claimed": [
41
+ "Under human supervision on an approved knowledge base, AI assistance can shorten handling time while quality is monitored.",
42
+ "Benefits are not uniform across staff: the most experienced agents need their own quality monitoring."
43
+ ],
21
44
  "what_cannot_be_claimed": [
22
45
  "Universal gains, autonomous deployment safety, privacy protection, reduced staffing requirements or educational learning gains."
23
46
  ],
47
+ "exceeds_evidence_boundary": [
48
+ "Claiming universal gains or that autonomous deployment is safe exceeds the boundary: no included study measures privacy incidents, local net cost or subgroup service quality."
49
+ ],
24
50
  "missing_evidence": [
25
- "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set."
51
+ "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set [no direct evidence]."
26
52
  ],
27
53
  "applicability": {
28
54
  "required_conditions": [
@@ -35,7 +61,7 @@
35
61
  "extensions": {
36
62
  "data_origin": "manual_curated",
37
63
  "benchmark_eligible": false,
38
- "note": "Conservative manual judgment about broad deployment; not a deterministic confidence-policy run or a probability.",
64
+ "note": "Confidence and the decision bound come from the deterministic policy in engine/decision_policy.py, enforced by the Pre-Verdict Gate; this record is a curated evidence selection, not a systematic review or a model run.",
39
65
  "knowledge_gaps": [
40
66
  {
41
67
  "gap_id": "G-001",
@@ -45,7 +71,7 @@
45
71
  "E-003",
46
72
  "E-004"
47
73
  ],
48
- "summary": "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set."
74
+ "summary": "Local net cost, privacy incident rates, subgroup service quality and sustained performance are not established by this selected evidence set [no direct evidence]."
49
75
  }
50
76
  ]
51
77
  }
@@ -87,7 +87,7 @@ AGENT_MCP_APPROVAL_REQUIRED = "AGENT_MCP_APPROVAL_REQUIRED"
87
87
  # skeptic requires a *different model family* than the primary analysis;
88
88
  # that can never be satisfied by spawning the same model in another session.
89
89
  ROLE_REQUIREMENTS: dict[str, dict[str, Any]] = {
90
- "education-planner": {
90
+ "research-planner": {
91
91
  "reasoning": "high", "speed": None, "cost": None,
92
92
  "structured_output": None, "context": None, "tool_use": None,
93
93
  "multimodal": None,
@@ -134,7 +134,7 @@ ROLE_REQUIREMENTS: dict[str, dict[str, Any]] = {
134
134
  # Human-readable one-line task per role (for the user-facing recommendation
135
135
  # table). Display metadata only — not a routing decision.
136
136
  ROLE_TASKS: dict[str, str] = {
137
- "education-planner": "Framing:把教学问题转成 EducationResearchFrame",
137
+ "research-planner": "Framing:把教学问题转成 EducationResearchFrame",
138
138
  "evidence-retriever": "检索支持与反方证据,去重初筛",
139
139
  "evidence-analyst": "证据结构化抽取为 Evidence Objects",
140
140
  "skeptic": "独立反证:9 项检查,找 null/negative/contradictory 证据",