eduevidence 6.0.0 → 6.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (189) hide show
  1. package/CONTRIBUTING.md +105 -0
  2. package/README.md +93 -38
  3. package/README.zh-CN.md +26 -6
  4. package/SKILL.md +11 -2
  5. package/assets/readme/landing-tour.gif +0 -0
  6. package/assets/readme/studio-tour.gif +0 -0
  7. package/bin/eduevidence.js +2 -1
  8. package/docs/architecture.md +319 -43
  9. package/docs/demo-workplace-ai.md +1 -1
  10. package/docs/install-guide.md +1 -1
  11. package/docs/orchestration-role-model.md +1 -1
  12. package/docs/release-closeout/README.md +1 -1
  13. package/docs/sciverse-api.md +125 -0
  14. package/eduevidence_cli.py +10 -0
  15. package/engine/decision_policy.py +96 -0
  16. package/engine/evidence_graph.py +14 -10
  17. package/engine/gaps.py +42 -22
  18. package/engine/ids.py +2 -0
  19. package/engine/library.py +6 -2
  20. package/engine/living.py +34 -4
  21. package/engine/migration.py +88 -3
  22. package/engine/orchestration.py +5 -5
  23. package/engine/paths.py +2 -0
  24. package/engine/pilot.py +34 -32
  25. package/engine/taxonomy.py +211 -0
  26. package/engine/tribunal.py +43 -31
  27. package/engine/versions.py +1 -1
  28. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1360 -146
  29. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +3 -3
  30. package/examples/ai-coding-assistant-evidence/citation_check.json +1 -1
  31. package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
  32. package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
  33. package/examples/ai-coding-assistant-evidence/report_spec.json +23 -12
  34. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +447 -127
  35. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +447 -127
  36. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +447 -127
  37. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +447 -127
  38. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +447 -127
  39. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1360 -146
  40. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1360 -146
  41. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1360 -146
  42. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1360 -146
  43. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1360 -146
  44. package/examples/ai-coding-assistant-evidence/result.json +13 -9
  45. package/examples/ai-coding-assistant-evidence/result.zh.json +45 -41
  46. package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
  47. package/examples/ai-coding-assistant-evidence/verdict.json +6 -2
  48. package/examples/spaced-retrieval-practice/applicability.json +14 -0
  49. package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
  50. package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
  51. package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
  52. package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
  53. package/examples/spaced-retrieval-practice/frame.json +58 -0
  54. package/examples/spaced-retrieval-practice/gate_report.json +101 -0
  55. package/examples/spaced-retrieval-practice/methodology.json +78 -0
  56. package/examples/spaced-retrieval-practice/report_spec.json +212 -0
  57. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
  58. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
  59. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
  60. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
  61. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
  62. package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
  63. package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
  64. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
  65. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
  66. package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
  67. package/examples/spaced-retrieval-practice/result.json +942 -0
  68. package/examples/spaced-retrieval-practice/result.zh.json +942 -0
  69. package/examples/spaced-retrieval-practice/skeptic.json +70 -0
  70. package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
  71. package/examples/spaced-retrieval-practice/verdict.json +93 -0
  72. package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
  73. package/examples/workplace-ai-assistant/claims.jsonl +4 -4
  74. package/examples/workplace-ai-assistant/evidence.jsonl +4 -4
  75. package/examples/workplace-ai-assistant/evidence_graph.json +15 -15
  76. package/examples/workplace-ai-assistant/final_verdict.json +78 -0
  77. package/examples/workplace-ai-assistant/gate_report.json +101 -0
  78. package/examples/workplace-ai-assistant/report_spec.json +209 -40
  79. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +435 -105
  80. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +435 -105
  81. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +435 -105
  82. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +435 -105
  83. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +435 -105
  84. package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
  85. package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
  86. package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
  87. package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
  88. package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
  89. package/examples/workplace-ai-assistant/result.json +82 -20
  90. package/examples/workplace-ai-assistant/result.zh.json +82 -20
  91. package/examples/workplace-ai-assistant/skeptic.json +72 -0
  92. package/examples/workplace-ai-assistant/verdict.json +36 -10
  93. package/integrations/agent_mcp.py +2 -2
  94. package/package.json +12 -3
  95. package/pyproject.toml +4 -3
  96. package/references/report-copy-style.md +67 -0
  97. package/references/retrieval-compliance.md +75 -0
  98. package/references/retrieval-protocol.md +20 -0
  99. package/retrieval/audit.py +27 -3
  100. package/retrieval/fetch.py +96 -0
  101. package/retrieval/sciverse.py +398 -0
  102. package/retrieval/search.py +47 -7
  103. package/schemas/applicability.schema.json +94 -0
  104. package/schemas/chart-spec.schema.json +10 -3
  105. package/schemas/evidence.schema.json +316 -43
  106. package/schemas/fetch-result.schema.json +2 -1
  107. package/schemas/report-result.schema.json +3 -3
  108. package/schemas/report-spec.schema.json +98 -100
  109. package/schemas/skeptic.schema.json +86 -0
  110. package/schemas/source.schema.json +21 -2
  111. package/schemas/v2/finding.schema.json +5 -1
  112. package/schemas/v2/methodology-audit.schema.json +5 -1
  113. package/schemas/v2/outcome.schema.json +28 -5
  114. package/schemas/v2/study.schema.json +5 -1
  115. package/schemas/vNext/autoevolve-session.schema.json +34 -1
  116. package/schemas/vNext/eval-snapshot.schema.json +77 -1
  117. package/schemas/vNext/execution-plan.schema.json +50 -1
  118. package/schemas/vNext/gap-priority.schema.json +54 -1
  119. package/schemas/vNext/negative-search-record.schema.json +68 -1
  120. package/schemas/vNext/research-iteration.schema.json +87 -1
  121. package/schemas/vNext/research-strategy.schema.json +62 -1
  122. package/schemas/vNext/skill-experiment.schema.json +90 -1
  123. package/schemas/vNext/task-spec.schema.json +156 -1
  124. package/schemas/vNext/worker-result.schema.json +60 -1
  125. package/schemas/verdict.schema.json +164 -28
  126. package/scripts/build_esl_artifacts.py +2 -2
  127. package/scripts/build_report_variants.py +18 -2
  128. package/scripts/build_result.py +74 -9
  129. package/scripts/check_package_parity.py +85 -0
  130. package/scripts/check_protocol_alignment.py +375 -0
  131. package/scripts/check_versioned_schemas.py +254 -0
  132. package/scripts/claim_audit.py +13 -8
  133. package/scripts/compute_confidence.py +10 -0
  134. package/scripts/did_regression.py +12 -2
  135. package/scripts/evidence_score.py +5 -2
  136. package/scripts/generate_new_projects.py +4 -4
  137. package/scripts/orchestrator.py +120 -24
  138. package/scripts/pre_verdict_gate.py +224 -26
  139. package/scripts/quickstart.py +18 -2
  140. package/scripts/run_workspace.py +7 -1
  141. package/scripts/skill_payload.py +4 -1
  142. package/scripts/test_adversarial_empirical.py +26 -19
  143. package/scripts/validate_schema.py +31 -1
  144. package/skill/agents/evaluation-designer.md +20 -4
  145. package/skill/agents/evidence-analyst.md +19 -3
  146. package/skill/agents/evidence-judge.md +50 -2
  147. package/skill/agents/evidence-retriever.md +20 -3
  148. package/skill/agents/intervention-designer.md +20 -4
  149. package/skill/agents/method-reviewer.md +18 -2
  150. package/skill/agents/{education-planner.md → research-planner.md} +19 -3
  151. package/skill/agents/skeptic.md +18 -2
  152. package/skill/roles/registry.yaml +11 -11
  153. package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
  154. package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
  155. package/skill/sub-skills/data-analysis/SKILL.md +34 -15
  156. package/skill/sub-skills/ethics-review/SKILL.md +33 -10
  157. package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
  158. package/skill/sub-skills/evidence-review/SKILL.md +31 -12
  159. package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
  160. package/skill/sub-skills/literature-review/SKILL.md +35 -14
  161. package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
  162. package/skill/sub-skills/report-generation/SKILL.md +28 -0
  163. package/skill/sub-skills/research-planning/SKILL.md +41 -14
  164. package/skill/sub-skills/study-design/SKILL.md +30 -9
  165. package/skill/task-briefs/adjudicate.md +32 -7
  166. package/skill/task-briefs/applicability.md +37 -2
  167. package/skill/task-briefs/audit.md +32 -7
  168. package/skill/task-briefs/challenge.md +34 -5
  169. package/skill/task-briefs/evaluate.md +30 -5
  170. package/skill/task-briefs/extract.md +31 -8
  171. package/skill/task-briefs/frame.md +39 -10
  172. package/skill/task-briefs/intervene.md +32 -6
  173. package/skill/task-briefs/present.md +32 -8
  174. package/skill/task-briefs/projection.md +36 -2
  175. package/skill/task-briefs/retrieve.md +36 -6
  176. package/skill/workflows/decision-and-pilot.md +76 -1
  177. package/skill/workflows/evaluate-and-update.md +83 -0
  178. package/skill/workflows/evidence-review.md +104 -0
  179. package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
  180. package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
  181. package/visualization/eduevidence-report/scripts/build_report.py +512 -65
  182. package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
  183. package/visualization/eduevidence-report/scripts/lieflat_engine.py +349 -38
  184. package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
  185. package/web/architecture.html +14885 -0
  186. package/web/studio/assets/index-B8tkF44Q.css +1 -0
  187. package/web/studio/index.html +2 -2
  188. package/web/studio/assets/index-CzXocaGv.css +0 -1
  189. /package/web/studio/assets/{index-pa7jD7n4.js → index-CQ6Keoyc.js} +0 -0
@@ -1 +1,90 @@
1
- {"$schema":"http://json-schema.org/draft-07/schema#","type":"object","required":["experiment_id","session_id","parent_skill_revision","hypothesis","mutation_scope","status"],"properties":{"experiment_id":{"type":"string"},"session_id":{"type":"string"},"parent_skill_revision":{"type":"string"},"hypothesis":{"type":"string","minLength":1},"mutation_scope":{"type":"array","items":{"type":"string"},"minItems":1},"changed_files":{"type":"array","items":{"type":"string"}},"candidate_commit":{"type":["string","null"]},"baseline_eval_id":{"type":["string","null"]},"candidate_eval_id":{"type":["string","null"]},"protected_hash_before":{"type":["string","null"]},"protected_hash_after":{"type":["string","null"]},"status":{"enum":["created","KEEP","REJECT","RETEST","HUMAN_REVIEW","CRASH","INVALID"]},"promotion_reason":{"type":"string"},"complexity_delta":{"type":"number"}},"additionalProperties":false}
1
+ {
2
+ "$schema": "http://json-schema.org/draft-07/schema#",
3
+ "title": "SkillExperiment",
4
+ "description": "One bounded self-evolution experiment (engine/autoevolve/core.py). Pairs a hypothesis and mutation scope with the baseline and candidate evaluation ids and the protected-tree hashes before and after, then records the promotion verdict. Automatic KEEP requires runner-owned isolation and an intact protected tree.",
5
+ "type": "object",
6
+ "required": [
7
+ "experiment_id",
8
+ "session_id",
9
+ "parent_skill_revision",
10
+ "hypothesis",
11
+ "mutation_scope",
12
+ "status"
13
+ ],
14
+ "properties": {
15
+ "experiment_id": {
16
+ "type": "string"
17
+ },
18
+ "session_id": {
19
+ "type": "string"
20
+ },
21
+ "parent_skill_revision": {
22
+ "type": "string"
23
+ },
24
+ "hypothesis": {
25
+ "type": "string",
26
+ "minLength": 1
27
+ },
28
+ "mutation_scope": {
29
+ "type": "array",
30
+ "items": {
31
+ "type": "string"
32
+ },
33
+ "minItems": 1
34
+ },
35
+ "changed_files": {
36
+ "type": "array",
37
+ "items": {
38
+ "type": "string"
39
+ }
40
+ },
41
+ "candidate_commit": {
42
+ "type": [
43
+ "string",
44
+ "null"
45
+ ]
46
+ },
47
+ "baseline_eval_id": {
48
+ "type": [
49
+ "string",
50
+ "null"
51
+ ]
52
+ },
53
+ "candidate_eval_id": {
54
+ "type": [
55
+ "string",
56
+ "null"
57
+ ]
58
+ },
59
+ "protected_hash_before": {
60
+ "type": [
61
+ "string",
62
+ "null"
63
+ ]
64
+ },
65
+ "protected_hash_after": {
66
+ "type": [
67
+ "string",
68
+ "null"
69
+ ]
70
+ },
71
+ "status": {
72
+ "enum": [
73
+ "created",
74
+ "KEEP",
75
+ "REJECT",
76
+ "RETEST",
77
+ "HUMAN_REVIEW",
78
+ "CRASH",
79
+ "INVALID"
80
+ ]
81
+ },
82
+ "promotion_reason": {
83
+ "type": "string"
84
+ },
85
+ "complexity_delta": {
86
+ "type": "number"
87
+ }
88
+ },
89
+ "additionalProperties": false
90
+ }
@@ -1 +1,156 @@
1
- {"$schema":"http://json-schema.org/draft-07/schema#","type":"object","required":["task_id","run_id","base_revision","stage","role","role_profile","objective","reason_for_delegation","evidence_axis","inputs","input_artifacts","allowed_capabilities","forbidden_actions","scope","budget","expected_outputs","output_contract","termination","execution_mode","independent","read_only","timeout_seconds","token_budget","metadata"],"properties":{"task_id":{"type":"string","minLength":1},"run_id":{"type":["string","null"]},"base_revision":{"type":["integer","null"],"minimum":0},"stage":{"enum":["frame","retrieve","extract","challenge","audit","adjudicate","applicability","intervene","evaluate"]},"role":{"type":"string"},"role_profile":{"type":["string","null"]},"objective":{"type":"string","minLength":1},"reason_for_delegation":{"type":["string","null"]},"evidence_axis":{"type":"string","minLength":1},"inputs":{"type":"array","items":{"type":"string"}},"input_artifacts":{"type":"array","items":{"type":"string"}},"allowed_capabilities":{"type":"array","items":{"type":"string"}},"forbidden_actions":{"type":"array","items":{"type":"string"}},"scope":{"type":"object"},"budget":{"type":"object"},"expected_outputs":{"type":"array","items":{"type":"string"}},"output_contract":{"type":"object"},"termination":{"type":"object"},"execution_mode":{"enum":["local","delegated"]},"independent":{"type":"boolean"},"read_only":{"type":"boolean"},"timeout_seconds":{"type":"integer","minimum":1},"token_budget":{"type":["integer","null"],"minimum":1},"metadata":{"type":"object"}},"additionalProperties":false}
1
+ {
2
+ "$schema": "http://json-schema.org/draft-07/schema#",
3
+ "title": "TaskSpec",
4
+ "description": "A dispatchable unit of work for one role (engine/orchestration.py TaskSpec). Fixes the stage, role, evidence axis, run and base revision, the granted capabilities, forbidden actions, budget, expected outputs and the output contract. Delegated specs are staging-only: canonical-state artifacts are refused by construction.",
5
+ "type": "object",
6
+ "required": [
7
+ "task_id",
8
+ "run_id",
9
+ "base_revision",
10
+ "stage",
11
+ "role",
12
+ "role_profile",
13
+ "objective",
14
+ "reason_for_delegation",
15
+ "evidence_axis",
16
+ "inputs",
17
+ "input_artifacts",
18
+ "allowed_capabilities",
19
+ "forbidden_actions",
20
+ "scope",
21
+ "budget",
22
+ "expected_outputs",
23
+ "output_contract",
24
+ "termination",
25
+ "execution_mode",
26
+ "independent",
27
+ "read_only",
28
+ "timeout_seconds",
29
+ "token_budget",
30
+ "metadata"
31
+ ],
32
+ "properties": {
33
+ "task_id": {
34
+ "type": "string",
35
+ "minLength": 1
36
+ },
37
+ "run_id": {
38
+ "type": [
39
+ "string",
40
+ "null"
41
+ ]
42
+ },
43
+ "base_revision": {
44
+ "type": [
45
+ "integer",
46
+ "null"
47
+ ],
48
+ "minimum": 0
49
+ },
50
+ "stage": {
51
+ "enum": [
52
+ "frame",
53
+ "retrieve",
54
+ "extract",
55
+ "challenge",
56
+ "audit",
57
+ "adjudicate",
58
+ "applicability",
59
+ "intervene",
60
+ "evaluate"
61
+ ]
62
+ },
63
+ "role": {
64
+ "type": "string"
65
+ },
66
+ "role_profile": {
67
+ "type": [
68
+ "string",
69
+ "null"
70
+ ]
71
+ },
72
+ "objective": {
73
+ "type": "string",
74
+ "minLength": 1
75
+ },
76
+ "reason_for_delegation": {
77
+ "type": [
78
+ "string",
79
+ "null"
80
+ ]
81
+ },
82
+ "evidence_axis": {
83
+ "type": "string",
84
+ "minLength": 1
85
+ },
86
+ "inputs": {
87
+ "type": "array",
88
+ "items": {
89
+ "type": "string"
90
+ }
91
+ },
92
+ "input_artifacts": {
93
+ "type": "array",
94
+ "items": {
95
+ "type": "string"
96
+ }
97
+ },
98
+ "allowed_capabilities": {
99
+ "type": "array",
100
+ "items": {
101
+ "type": "string"
102
+ }
103
+ },
104
+ "forbidden_actions": {
105
+ "type": "array",
106
+ "items": {
107
+ "type": "string"
108
+ }
109
+ },
110
+ "scope": {
111
+ "type": "object"
112
+ },
113
+ "budget": {
114
+ "type": "object"
115
+ },
116
+ "expected_outputs": {
117
+ "type": "array",
118
+ "items": {
119
+ "type": "string"
120
+ }
121
+ },
122
+ "output_contract": {
123
+ "type": "object"
124
+ },
125
+ "termination": {
126
+ "type": "object"
127
+ },
128
+ "execution_mode": {
129
+ "enum": [
130
+ "local",
131
+ "delegated"
132
+ ]
133
+ },
134
+ "independent": {
135
+ "type": "boolean"
136
+ },
137
+ "read_only": {
138
+ "type": "boolean"
139
+ },
140
+ "timeout_seconds": {
141
+ "type": "integer",
142
+ "minimum": 1
143
+ },
144
+ "token_budget": {
145
+ "type": [
146
+ "integer",
147
+ "null"
148
+ ],
149
+ "minimum": 1
150
+ },
151
+ "metadata": {
152
+ "type": "object"
153
+ }
154
+ },
155
+ "additionalProperties": false
156
+ }
@@ -1 +1,60 @@
1
- {"$schema":"http://json-schema.org/draft-07/schema#","type":"object","required":["task_id","status","staging_artifacts","validated","validation_issues","metrics","summary"],"properties":{"task_id":{"type":"string","minLength":1},"status":{"enum":["completed","failed","blocked"]},"staging_artifacts":{"type":"array","items":{"type":"object","required":["artifact_type"],"properties":{"artifact_type":{"type":"string","minLength":1}},"additionalProperties":true}},"validated":{"type":"boolean"},"validation_issues":{"type":"array","items":{"type":"string"}},"metrics":{"type":"object"},"summary":{"type":"string"}},"additionalProperties":false}
1
+ {
2
+ "$schema": "http://json-schema.org/draft-07/schema#",
3
+ "title": "WorkerResult",
4
+ "description": "The lead process's verdict on one worker's submission (engine/worker_result.py). validate_worker_output() reconstructs this from untrusted worker output: any worker-supplied validated flag is ignored, canonical-state artifacts are rejected, and only results that pass here may enter a Judge context.",
5
+ "type": "object",
6
+ "required": [
7
+ "task_id",
8
+ "status",
9
+ "staging_artifacts",
10
+ "validated",
11
+ "validation_issues",
12
+ "metrics",
13
+ "summary"
14
+ ],
15
+ "properties": {
16
+ "task_id": {
17
+ "type": "string",
18
+ "minLength": 1
19
+ },
20
+ "status": {
21
+ "enum": [
22
+ "completed",
23
+ "failed",
24
+ "blocked"
25
+ ]
26
+ },
27
+ "staging_artifacts": {
28
+ "type": "array",
29
+ "items": {
30
+ "type": "object",
31
+ "required": [
32
+ "artifact_type"
33
+ ],
34
+ "properties": {
35
+ "artifact_type": {
36
+ "type": "string",
37
+ "minLength": 1
38
+ }
39
+ },
40
+ "additionalProperties": true
41
+ }
42
+ },
43
+ "validated": {
44
+ "type": "boolean"
45
+ },
46
+ "validation_issues": {
47
+ "type": "array",
48
+ "items": {
49
+ "type": "string"
50
+ }
51
+ },
52
+ "metrics": {
53
+ "type": "object"
54
+ },
55
+ "summary": {
56
+ "type": "string"
57
+ }
58
+ },
59
+ "additionalProperties": false
60
+ }
@@ -5,52 +5,188 @@
5
5
  "description": "Output of the Evidence Tribunal: what the evidence supports, cannot support, and the recommended action. Decision is one of ADOPT / PILOT / REJECT / INSUFFICIENT EVIDENCE (expressed here as adopt|pilot|reject|insufficient_evidence). confidence 由 scripts/compute_confidence.py 确定性计算并覆盖模型值;confidence_score 是规则化指数(0-1),不是概率。扩展字段一律放在 extensions 内。",
6
6
  "type": "object",
7
7
  "additionalProperties": false,
8
- "required": ["decision_question", "recommended_action", "confidence"],
8
+ "required": [
9
+ "decision_question",
10
+ "recommended_action",
11
+ "confidence"
12
+ ],
9
13
  "properties": {
10
- "decision_question": { "type": "string" },
11
- "target_population": { "type": "string" },
12
- "target_context": { "type": "string" },
13
- "supported_claims": { "type": "array", "items": { "type": "string" } },
14
- "uncertain_claims": { "type": "array", "items": { "type": "string" } },
15
- "contradicted_claims": { "type": "array", "items": { "type": "string" } },
16
- "reason_for_disagreement": { "type": "string" },
17
- "methodology_summary": { "type": "string" },
18
- "outcome_specific_findings": { "type": "object", "additionalProperties": true },
19
- "short_term_effect": { "type": ["string", "null"] },
20
- "long_term_effect": { "type": ["string", "null"] },
21
- "transfer_effect": { "type": ["string", "null"] },
22
- "risk_effect": { "type": ["string", "null"] },
23
- "applicability": { "type": "object", "additionalProperties": true },
24
- "confidence": { "type": "string", "enum": ["High", "Moderate", "Low", "Insufficient"] },
14
+ "decision_question": {
15
+ "type": "string"
16
+ },
17
+ "target_population": {
18
+ "type": "string"
19
+ },
20
+ "target_context": {
21
+ "type": "string"
22
+ },
23
+ "supported_claims": {
24
+ "type": "array",
25
+ "items": {
26
+ "type": "string"
27
+ }
28
+ },
29
+ "uncertain_claims": {
30
+ "type": "array",
31
+ "items": {
32
+ "type": "string"
33
+ }
34
+ },
35
+ "contradicted_claims": {
36
+ "type": "array",
37
+ "items": {
38
+ "type": "string"
39
+ }
40
+ },
41
+ "reason_for_disagreement": {
42
+ "type": "string"
43
+ },
44
+ "methodology_summary": {
45
+ "type": "string"
46
+ },
47
+ "outcome_specific_findings": {
48
+ "type": "object",
49
+ "additionalProperties": true
50
+ },
51
+ "short_term_effect": {
52
+ "type": [
53
+ "string",
54
+ "null"
55
+ ]
56
+ },
57
+ "long_term_effect": {
58
+ "type": [
59
+ "string",
60
+ "null"
61
+ ]
62
+ },
63
+ "transfer_effect": {
64
+ "type": [
65
+ "string",
66
+ "null"
67
+ ]
68
+ },
69
+ "risk_effect": {
70
+ "type": [
71
+ "string",
72
+ "null"
73
+ ]
74
+ },
75
+ "applicability": {
76
+ "type": "object",
77
+ "additionalProperties": true
78
+ },
79
+ "confidence": {
80
+ "type": "string",
81
+ "enum": [
82
+ "High",
83
+ "Moderate",
84
+ "Low",
85
+ "Insufficient"
86
+ ]
87
+ },
25
88
  "confidence_score": {
26
- "type": ["number", "null"],
89
+ "type": [
90
+ "number",
91
+ "null"
92
+ ],
27
93
  "minimum": 0,
28
94
  "maximum": 1,
29
95
  "description": "规则化置信度指数(0-1),由 compute_confidence.py 覆盖模型值。不是概率,禁止宣传为百分比。"
30
96
  },
31
- "confidence_policy_version": { "type": "string", "description": "确定性置信度策略版本号(如 2026-08-12.v1)。" },
32
- "independent_studies": { "type": ["integer", "null"], "minimum": 0, "description": "独立研究数(按 study_id/source_id 去重)。" },
33
- "independent_samples": { "type": ["integer", "null"], "minimum": 0, "description": "独立样本数(按 sample_id 去重)。" },
97
+ "confidence_policy_version": {
98
+ "type": "string",
99
+ "description": "确定性置信度策略版本号(如 2026-08-12.v1)。"
100
+ },
101
+ "independent_studies": {
102
+ "type": [
103
+ "integer",
104
+ "null"
105
+ ],
106
+ "minimum": 0,
107
+ "description": "独立研究数(按 study_id/source_id 去重)。"
108
+ },
109
+ "independent_samples": {
110
+ "type": [
111
+ "integer",
112
+ "null"
113
+ ],
114
+ "minimum": 0,
115
+ "description": "独立样本数(按 sample_id 去重)。"
116
+ },
34
117
  "confidence_breakdown": {
35
118
  "type": "object",
36
119
  "additionalProperties": true,
37
120
  "description": "Rule-based components: evidence_quality, consistency, directness, evidence_count, independent_studies, independent_samples, conflict_penalty, unsupported_penalty."
38
121
  },
39
122
  "raw_model_confidence": {
40
- "type": ["string", "null"],
123
+ "type": [
124
+ "string",
125
+ "null"
126
+ ],
41
127
  "description": "模型原始 confidence 输出(被确定性值覆盖前的值,仅供审计比对)。"
42
128
  },
43
- "raw_model_confidence_breakdown": { "type": "object", "additionalProperties": true },
44
- "what_can_be_claimed": { "type": "array", "items": { "type": "string" } },
45
- "what_cannot_be_claimed": { "type": "array", "items": { "type": "string" } },
46
- "missing_evidence": { "type": "array", "items": { "type": "string" } },
47
- "recommended_action": { "type": "string", "enum": ["adopt", "pilot", "reject", "insufficient_evidence"] },
48
- "decision_rationale": { "type": "string" },
49
- "exceeds_evidence_boundary": { "type": "array", "items": { "type": "string" }, "description": "Conclusions that currently go beyond the evidence boundary." },
129
+ "raw_model_confidence_breakdown": {
130
+ "type": "object",
131
+ "additionalProperties": true
132
+ },
133
+ "what_can_be_claimed": {
134
+ "type": "array",
135
+ "items": {
136
+ "type": "string"
137
+ }
138
+ },
139
+ "what_cannot_be_claimed": {
140
+ "type": "array",
141
+ "items": {
142
+ "type": "string"
143
+ }
144
+ },
145
+ "missing_evidence": {
146
+ "type": "array",
147
+ "items": {
148
+ "type": "string"
149
+ }
150
+ },
151
+ "recommended_action": {
152
+ "type": "string",
153
+ "enum": [
154
+ "adopt",
155
+ "pilot",
156
+ "reject",
157
+ "insufficient_evidence"
158
+ ]
159
+ },
160
+ "decision_rationale": {
161
+ "type": "string"
162
+ },
163
+ "exceeds_evidence_boundary": {
164
+ "type": "array",
165
+ "items": {
166
+ "type": "string"
167
+ },
168
+ "description": "Conclusions that currently go beyond the evidence boundary."
169
+ },
50
170
  "extensions": {
51
171
  "type": "object",
52
172
  "description": "结构化扩展字段的统一容器(P1-01)。未列入本 schema 的字段必须放在这里,禁止在顶层新增属性。",
53
173
  "additionalProperties": true
174
+ },
175
+ "strongest_support": {
176
+ "type": "string",
177
+ "description": "The single strongest conclusion the evidence supports, as a complete reader-facing sentence (<=60 chars zh / ~15 words en). Written by the adjudicator, not assembled by the renderer."
178
+ },
179
+ "key_uncertainty": {
180
+ "type": "string",
181
+ "description": "The decision-relevant uncertainty or counter-evidence, as a complete reader-facing sentence (<=70 chars zh / ~18 words en)."
182
+ },
183
+ "main_risk": {
184
+ "type": "string",
185
+ "description": "The principal risk of acting, as a complete reader-facing sentence (<=60 chars zh / ~15 words en)."
186
+ },
187
+ "next_action": {
188
+ "type": "string",
189
+ "description": "The recommended next step, as a complete reader-facing sentence (<=80 chars zh / ~20 words en)."
54
190
  }
55
191
  }
56
192
  }
@@ -1552,7 +1552,7 @@ result_en = {
1552
1552
  "complexity": "L",
1553
1553
  "mode": "agent_mcp_enhanced",
1554
1554
  "agents": [
1555
- "education-planner",
1555
+ "research-planner",
1556
1556
  "evidence-retriever",
1557
1557
  "evidence-analyst",
1558
1558
  "skeptic",
@@ -1651,7 +1651,7 @@ result_zh = {
1651
1651
  "complexity": "L",
1652
1652
  "mode": "agent_mcp_enhanced",
1653
1653
  "agents": [
1654
- "education-planner",
1654
+ "research-planner",
1655
1655
  "evidence-retriever",
1656
1656
  "evidence-analyst",
1657
1657
  "skeptic",
@@ -40,7 +40,9 @@ def bake(examples: Path, *, force: bool = False) -> list[dict]:
40
40
  if not force and manifest.is_file():
41
41
  try:
42
42
  prior = json.loads(manifest.read_text(encoding='utf-8'))
43
- valid = prior.get('cache_key') == cache_key and all(
43
+ main = prior.get('main_report') or {}
44
+ main_ok = bool(main.get('file')) and (directory / main['file']).is_file() and hashlib.sha256((directory / main['file']).read_bytes()).hexdigest() == main.get('sha256')
45
+ valid = prior.get('cache_key') == cache_key and main_ok and all(
44
46
  (out_dir / record['file']).is_file() and hashlib.sha256((out_dir / record['file']).read_bytes()).hexdigest() == record['sha256']
45
47
  for record in prior.get('reports', [])) and len(prior.get('reports', [])) == len(THEMES)
46
48
  if valid:
@@ -64,8 +66,22 @@ def bake(examples: Path, *, force: bool = False) -> list[dict]:
64
66
  raise RuntimeError(f'{directory.name}/{theme}: renderer rejected input\n{completed.stdout}\n{completed.stderr}')
65
67
  os.replace(temporary, target)
66
68
  records.append({'theme': theme, 'file': target.name, 'sha256': hashlib.sha256(target.read_bytes()).hexdigest()})
69
+ # Also refresh the pack-root report. This used to write only the themed
70
+ # variants, so examples/*/EduEvidence_Report.html kept whatever bytes it
71
+ # was first rendered with - the packaged example shipped a report the
72
+ # current renderer would not produce.
73
+ from_default = next((r for r in records if r['theme'] == 'claude'), records[0])
74
+ main_target = directory / 'EduEvidence_Report.html'
75
+ main_temp = directory / '.EduEvidence_Report.pending.html'
76
+ main_temp.write_bytes((out_dir / from_default['file']).read_bytes())
77
+ os.replace(main_temp, main_target)
78
+ main_sha = hashlib.sha256(main_target.read_bytes()).hexdigest()
79
+
67
80
  value = {'schema_version': 1, 'project': directory.name, 'cache_key': cache_key,
68
- 'result_sha256': result_hash, 'renderer_sha256': engine_hash, 'reports': records}
81
+ 'result_sha256': result_hash, 'renderer_sha256': engine_hash,
82
+ 'main_report': {'file': main_target.name, 'sha256': main_sha,
83
+ 'theme': from_default['theme']},
84
+ 'reports': records}
69
85
  manifest.write_text(json.dumps(value, indent=2) + '\n', encoding='utf-8')
70
86
  reports.append(value)
71
87
  print(f'{directory.name}: {len(records)} verified report variants')