eduevidence 6.0.0 → 6.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (267) hide show
  1. package/CHANGELOG.md +395 -0
  2. package/CONTRIBUTING.md +105 -0
  3. package/README.md +113 -49
  4. package/README.zh-CN.md +39 -12
  5. package/SKILL.md +15 -5
  6. package/assets/readme/landing-tour.gif +0 -0
  7. package/assets/readme/studio-tour.gif +0 -0
  8. package/benchmarks/evidence-library.json +277 -1
  9. package/bin/eduevidence.js +2 -1
  10. package/docs/architecture.md +325 -46
  11. package/docs/demo-workplace-ai.md +1 -1
  12. package/docs/install-guide.md +1 -1
  13. package/docs/j-ev-experimental.md +250 -0
  14. package/docs/orchestration-role-model.md +1 -1
  15. package/docs/release-closeout/README.md +1 -1
  16. package/docs/reproducibility.md +138 -0
  17. package/docs/sciverse-api.md +125 -0
  18. package/domains/_neutral/copy/few_shots.json +21 -0
  19. package/domains/_neutral/copy/framing_lexicon.json +19 -0
  20. package/domains/_neutral/copy/module_labels.json +5 -0
  21. package/domains/_neutral/copy/module_labels_footer.json +102 -0
  22. package/domains/_neutral/copy/module_labels_modules.json +204 -0
  23. package/domains/_neutral/copy/module_labels_nav.json +126 -0
  24. package/domains/_neutral/copy/module_labels_summary.json +98 -0
  25. package/domains/_neutral/copy/module_labels_tables.json +164 -0
  26. package/domains/_neutral/copy/module_labels_v2.json +90 -0
  27. package/domains/_neutral/copy/risk_constructs.json +20 -0
  28. package/domains/_neutral/copy/section_titles.json +66 -0
  29. package/domains/_neutral/copy/terminology.json +11 -0
  30. package/domains/check_copy_packs.py +103 -0
  31. package/domains/education/copy/few_shots.json +22 -0
  32. package/domains/education/copy/framing_enums.json +167 -0
  33. package/domains/education/copy/framing_lexicon.json +166 -0
  34. package/domains/education/copy/module_labels.json +169 -0
  35. package/domains/education/copy/risk_constructs.json +48 -0
  36. package/domains/education/copy/section_titles.json +186 -0
  37. package/domains/education/copy/terminology.json +70 -0
  38. package/domains/education/manifest.json +1 -1
  39. package/domains/education/outcome_taxonomy.json +2 -2
  40. package/domains/manifest.json +1 -1
  41. package/domains/policy/copy/few_shots.json +22 -0
  42. package/domains/policy/copy/framing_enums.json +94 -0
  43. package/domains/policy/copy/framing_lexicon.json +174 -0
  44. package/domains/policy/copy/module_labels.json +168 -0
  45. package/domains/policy/copy/risk_constructs.json +33 -0
  46. package/domains/policy/copy/section_titles.json +186 -0
  47. package/domains/policy/copy/terminology.json +64 -0
  48. package/eduevidence_cli.py +10 -0
  49. package/engine/capabilities.py +57 -5
  50. package/engine/decision_policy.py +167 -0
  51. package/engine/evidence_graph.py +14 -10
  52. package/engine/gaps.py +42 -22
  53. package/engine/ids.py +2 -0
  54. package/engine/library.py +6 -2
  55. package/engine/library_builtin.py +7 -4
  56. package/engine/living.py +34 -4
  57. package/engine/migration.py +88 -3
  58. package/engine/orchestration.py +5 -5
  59. package/engine/paths.py +2 -0
  60. package/engine/pilot.py +34 -32
  61. package/engine/taxonomy.py +211 -0
  62. package/engine/tribunal.py +49 -43
  63. package/engine/versions.py +1 -1
  64. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1361 -147
  65. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +3 -3
  66. package/examples/ai-coding-assistant-evidence/citation_check.json +1 -1
  67. package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
  68. package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
  69. package/examples/ai-coding-assistant-evidence/report_spec.json +23 -12
  70. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +448 -128
  71. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +448 -128
  72. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +448 -128
  73. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +448 -128
  74. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +448 -128
  75. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1360 -146
  76. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1360 -146
  77. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1360 -146
  78. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1360 -146
  79. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1360 -146
  80. package/examples/ai-coding-assistant-evidence/result.json +13 -9
  81. package/examples/ai-coding-assistant-evidence/result.zh.json +45 -41
  82. package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
  83. package/examples/ai-coding-assistant-evidence/verdict.json +6 -2
  84. package/examples/spaced-retrieval-practice/EduEvidence_Report.html +2728 -0
  85. package/examples/spaced-retrieval-practice/applicability.json +14 -0
  86. package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
  87. package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
  88. package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
  89. package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
  90. package/examples/spaced-retrieval-practice/frame.json +58 -0
  91. package/examples/spaced-retrieval-practice/gate_report.json +101 -0
  92. package/examples/spaced-retrieval-practice/methodology.json +78 -0
  93. package/examples/spaced-retrieval-practice/report.html +2522 -0
  94. package/examples/spaced-retrieval-practice/report_spec.json +212 -0
  95. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
  96. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
  97. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
  98. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
  99. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
  100. package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
  101. package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
  102. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
  103. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
  104. package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
  105. package/examples/spaced-retrieval-practice/result.json +942 -0
  106. package/examples/spaced-retrieval-practice/result.zh.json +942 -0
  107. package/examples/spaced-retrieval-practice/skeptic.json +70 -0
  108. package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
  109. package/examples/spaced-retrieval-practice/verdict.json +93 -0
  110. package/examples/workplace-ai-assistant/EduEvidence_Report.html +2814 -0
  111. package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
  112. package/examples/workplace-ai-assistant/claims.jsonl +4 -4
  113. package/examples/workplace-ai-assistant/evidence.jsonl +4 -4
  114. package/examples/workplace-ai-assistant/evidence_graph.json +15 -15
  115. package/examples/workplace-ai-assistant/final_verdict.json +78 -0
  116. package/examples/workplace-ai-assistant/gate_report.json +101 -0
  117. package/examples/workplace-ai-assistant/report_spec.json +209 -40
  118. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +449 -119
  119. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +449 -119
  120. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +449 -119
  121. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +449 -119
  122. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +449 -119
  123. package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
  124. package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
  125. package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
  126. package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
  127. package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
  128. package/examples/workplace-ai-assistant/result.json +82 -20
  129. package/examples/workplace-ai-assistant/result.zh.json +82 -20
  130. package/examples/workplace-ai-assistant/skeptic.json +72 -0
  131. package/examples/workplace-ai-assistant/verdict.json +36 -10
  132. package/integrations/agent_mcp.py +2 -2
  133. package/integrations/jev/__init__.py +115 -0
  134. package/integrations/jev/approval.py +212 -0
  135. package/integrations/jev/cli.py +84 -0
  136. package/integrations/jev/config.py +112 -0
  137. package/integrations/jev/gateway.py +128 -0
  138. package/integrations/jev/modes.py +38 -0
  139. package/integrations/jev/tools_classify.py +88 -0
  140. package/integrations/jev/tools_extract.py +111 -0
  141. package/integrations/jev/tools_rerank.py +71 -0
  142. package/integrations/jev/tools_screen.py +87 -0
  143. package/integrations/jev/tools_verify.py +95 -0
  144. package/integrations/jev_mcp.py +22 -0
  145. package/integrations/semantic_decide.py +286 -0
  146. package/integrations/semdecide_cli.py +55 -0
  147. package/package.json +19 -2
  148. package/pyproject.toml +4 -3
  149. package/references/report-copy-style.md +107 -0
  150. package/references/retrieval-compliance.md +75 -0
  151. package/references/retrieval-protocol.md +20 -0
  152. package/retrieval/audit.py +27 -3
  153. package/retrieval/fetch.py +96 -0
  154. package/retrieval/sciverse.py +398 -0
  155. package/retrieval/search.py +47 -7
  156. package/schemas/applicability.schema.json +94 -0
  157. package/schemas/chart-spec.schema.json +10 -3
  158. package/schemas/evidence.schema.json +316 -43
  159. package/schemas/fetch-result.schema.json +2 -1
  160. package/schemas/report-result.schema.json +3 -3
  161. package/schemas/report-spec.schema.json +98 -100
  162. package/schemas/skeptic.schema.json +86 -0
  163. package/schemas/source.schema.json +21 -2
  164. package/schemas/v2/decision-snapshot.schema.json +20 -9
  165. package/schemas/v2/finding.schema.json +5 -1
  166. package/schemas/v2/intake.schema.json +191 -0
  167. package/schemas/v2/methodology-audit.schema.json +5 -1
  168. package/schemas/v2/outcome.schema.json +28 -5
  169. package/schemas/v2/study.schema.json +5 -1
  170. package/schemas/vNext/autoevolve-session.schema.json +34 -1
  171. package/schemas/vNext/eval-snapshot.schema.json +77 -1
  172. package/schemas/vNext/execution-plan.schema.json +50 -1
  173. package/schemas/vNext/gap-priority.schema.json +54 -1
  174. package/schemas/vNext/negative-search-record.schema.json +68 -1
  175. package/schemas/vNext/research-iteration.schema.json +87 -1
  176. package/schemas/vNext/research-strategy.schema.json +62 -1
  177. package/schemas/vNext/skill-experiment.schema.json +90 -1
  178. package/schemas/vNext/task-spec.schema.json +156 -1
  179. package/schemas/vNext/worker-result.schema.json +60 -1
  180. package/schemas/verdict.schema.json +164 -28
  181. package/scripts/build_evidence_library.py +15 -5
  182. package/scripts/build_report_variants.py +18 -2
  183. package/scripts/build_result.py +74 -9
  184. package/scripts/check_package_parity.py +85 -0
  185. package/scripts/check_protocol_alignment.py +375 -0
  186. package/scripts/check_versioned_schemas.py +254 -0
  187. package/scripts/claim_audit.py +13 -8
  188. package/scripts/compute_confidence.py +10 -0
  189. package/scripts/dashboard_server.py +13 -2
  190. package/scripts/did_regression.py +12 -2
  191. package/scripts/evidence_score.py +5 -2
  192. package/scripts/intake/__init__.py +31 -0
  193. package/scripts/intake/__main__.py +18 -0
  194. package/scripts/intake/background.py +78 -0
  195. package/scripts/intake/browser.py +79 -0
  196. package/scripts/intake/cli.py +57 -0
  197. package/scripts/intake/constants.py +57 -0
  198. package/scripts/intake/depth.py +53 -0
  199. package/scripts/intake/enhancements.py +106 -0
  200. package/scripts/intake/hooks.py +90 -0
  201. package/scripts/intake/prefs.py +76 -0
  202. package/scripts/intake/prompts.py +85 -0
  203. package/scripts/intake/session.py +152 -0
  204. package/scripts/lint_file_layers.py +126 -0
  205. package/scripts/orchestrator.py +187 -40
  206. package/scripts/pre_verdict_gate.py +241 -29
  207. package/scripts/quickstart.py +18 -2
  208. package/scripts/run_workspace.py +7 -1
  209. package/scripts/skill_lint.py +11 -1
  210. package/scripts/skill_payload.py +6 -3
  211. package/scripts/test_adversarial_empirical.py +96 -25
  212. package/scripts/validate_schema.py +31 -1
  213. package/skill/agents/evaluation-designer.md +20 -4
  214. package/skill/agents/evidence-analyst.md +19 -3
  215. package/skill/agents/evidence-judge.md +98 -8
  216. package/skill/agents/evidence-retriever.md +20 -3
  217. package/skill/agents/intervention-designer.md +20 -4
  218. package/skill/agents/method-reviewer.md +18 -2
  219. package/skill/agents/{education-planner.md → research-planner.md} +19 -3
  220. package/skill/agents/skeptic.md +18 -2
  221. package/skill/roles/registry.yaml +11 -11
  222. package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
  223. package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
  224. package/skill/sub-skills/data-analysis/SKILL.md +34 -15
  225. package/skill/sub-skills/ethics-review/SKILL.md +33 -10
  226. package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
  227. package/skill/sub-skills/evidence-review/SKILL.md +31 -12
  228. package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
  229. package/skill/sub-skills/literature-review/SKILL.md +35 -14
  230. package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
  231. package/skill/sub-skills/report-generation/SKILL.md +28 -0
  232. package/skill/sub-skills/research-planning/SKILL.md +41 -14
  233. package/skill/sub-skills/study-design/SKILL.md +30 -9
  234. package/skill/task-briefs/adjudicate.md +32 -7
  235. package/skill/task-briefs/applicability.md +37 -2
  236. package/skill/task-briefs/audit.md +32 -7
  237. package/skill/task-briefs/challenge.md +34 -5
  238. package/skill/task-briefs/evaluate.md +30 -5
  239. package/skill/task-briefs/extract.md +31 -8
  240. package/skill/task-briefs/frame.md +39 -10
  241. package/skill/task-briefs/intervene.md +32 -6
  242. package/skill/task-briefs/present.md +32 -8
  243. package/skill/task-briefs/projection.md +36 -2
  244. package/skill/task-briefs/retrieve.md +36 -6
  245. package/skill/workflows/decision-and-pilot.md +76 -1
  246. package/skill/workflows/evaluate-and-update.md +83 -0
  247. package/skill/workflows/evidence-review.md +104 -0
  248. package/skill/workflows/experimental-jev.md +170 -0
  249. package/skill/workflows/intake.md +120 -0
  250. package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
  251. package/visualization/eduevidence-report/scripts/build_infographics.py +37 -15
  252. package/visualization/eduevidence-report/scripts/build_report.py +435 -575
  253. package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
  254. package/visualization/eduevidence-report/scripts/lieflat_engine.py +349 -38
  255. package/visualization/eduevidence-report/scripts/report_copy_pack.py +296 -0
  256. package/visualization/eduevidence-report/scripts/report_copy_policy_guard.py +47 -0
  257. package/visualization/eduevidence-report/scripts/zh_labels.py +141 -1
  258. package/web/architecture.html +14885 -0
  259. package/web/studio/assets/index-B8tkF44Q.css +1 -0
  260. package/web/studio/index.html +2 -2
  261. package/scripts/build_esl_artifacts.py +0 -1921
  262. package/scripts/build_killer_demo.py +0 -295
  263. package/scripts/enrich_projects_human_and_lieflat.py +0 -315
  264. package/scripts/generate_new_projects.py +0 -686
  265. package/scripts/sync_killer_demo_report.py +0 -270
  266. package/web/studio/assets/index-CzXocaGv.css +0 -1
  267. /package/web/studio/assets/{index-pa7jD7n4.js → index-CQ6Keoyc.js} +0 -0
@@ -1 +1,156 @@
1
- {"$schema":"http://json-schema.org/draft-07/schema#","type":"object","required":["task_id","run_id","base_revision","stage","role","role_profile","objective","reason_for_delegation","evidence_axis","inputs","input_artifacts","allowed_capabilities","forbidden_actions","scope","budget","expected_outputs","output_contract","termination","execution_mode","independent","read_only","timeout_seconds","token_budget","metadata"],"properties":{"task_id":{"type":"string","minLength":1},"run_id":{"type":["string","null"]},"base_revision":{"type":["integer","null"],"minimum":0},"stage":{"enum":["frame","retrieve","extract","challenge","audit","adjudicate","applicability","intervene","evaluate"]},"role":{"type":"string"},"role_profile":{"type":["string","null"]},"objective":{"type":"string","minLength":1},"reason_for_delegation":{"type":["string","null"]},"evidence_axis":{"type":"string","minLength":1},"inputs":{"type":"array","items":{"type":"string"}},"input_artifacts":{"type":"array","items":{"type":"string"}},"allowed_capabilities":{"type":"array","items":{"type":"string"}},"forbidden_actions":{"type":"array","items":{"type":"string"}},"scope":{"type":"object"},"budget":{"type":"object"},"expected_outputs":{"type":"array","items":{"type":"string"}},"output_contract":{"type":"object"},"termination":{"type":"object"},"execution_mode":{"enum":["local","delegated"]},"independent":{"type":"boolean"},"read_only":{"type":"boolean"},"timeout_seconds":{"type":"integer","minimum":1},"token_budget":{"type":["integer","null"],"minimum":1},"metadata":{"type":"object"}},"additionalProperties":false}
1
+ {
2
+ "$schema": "http://json-schema.org/draft-07/schema#",
3
+ "title": "TaskSpec",
4
+ "description": "A dispatchable unit of work for one role (engine/orchestration.py TaskSpec). Fixes the stage, role, evidence axis, run and base revision, the granted capabilities, forbidden actions, budget, expected outputs and the output contract. Delegated specs are staging-only: canonical-state artifacts are refused by construction.",
5
+ "type": "object",
6
+ "required": [
7
+ "task_id",
8
+ "run_id",
9
+ "base_revision",
10
+ "stage",
11
+ "role",
12
+ "role_profile",
13
+ "objective",
14
+ "reason_for_delegation",
15
+ "evidence_axis",
16
+ "inputs",
17
+ "input_artifacts",
18
+ "allowed_capabilities",
19
+ "forbidden_actions",
20
+ "scope",
21
+ "budget",
22
+ "expected_outputs",
23
+ "output_contract",
24
+ "termination",
25
+ "execution_mode",
26
+ "independent",
27
+ "read_only",
28
+ "timeout_seconds",
29
+ "token_budget",
30
+ "metadata"
31
+ ],
32
+ "properties": {
33
+ "task_id": {
34
+ "type": "string",
35
+ "minLength": 1
36
+ },
37
+ "run_id": {
38
+ "type": [
39
+ "string",
40
+ "null"
41
+ ]
42
+ },
43
+ "base_revision": {
44
+ "type": [
45
+ "integer",
46
+ "null"
47
+ ],
48
+ "minimum": 0
49
+ },
50
+ "stage": {
51
+ "enum": [
52
+ "frame",
53
+ "retrieve",
54
+ "extract",
55
+ "challenge",
56
+ "audit",
57
+ "adjudicate",
58
+ "applicability",
59
+ "intervene",
60
+ "evaluate"
61
+ ]
62
+ },
63
+ "role": {
64
+ "type": "string"
65
+ },
66
+ "role_profile": {
67
+ "type": [
68
+ "string",
69
+ "null"
70
+ ]
71
+ },
72
+ "objective": {
73
+ "type": "string",
74
+ "minLength": 1
75
+ },
76
+ "reason_for_delegation": {
77
+ "type": [
78
+ "string",
79
+ "null"
80
+ ]
81
+ },
82
+ "evidence_axis": {
83
+ "type": "string",
84
+ "minLength": 1
85
+ },
86
+ "inputs": {
87
+ "type": "array",
88
+ "items": {
89
+ "type": "string"
90
+ }
91
+ },
92
+ "input_artifacts": {
93
+ "type": "array",
94
+ "items": {
95
+ "type": "string"
96
+ }
97
+ },
98
+ "allowed_capabilities": {
99
+ "type": "array",
100
+ "items": {
101
+ "type": "string"
102
+ }
103
+ },
104
+ "forbidden_actions": {
105
+ "type": "array",
106
+ "items": {
107
+ "type": "string"
108
+ }
109
+ },
110
+ "scope": {
111
+ "type": "object"
112
+ },
113
+ "budget": {
114
+ "type": "object"
115
+ },
116
+ "expected_outputs": {
117
+ "type": "array",
118
+ "items": {
119
+ "type": "string"
120
+ }
121
+ },
122
+ "output_contract": {
123
+ "type": "object"
124
+ },
125
+ "termination": {
126
+ "type": "object"
127
+ },
128
+ "execution_mode": {
129
+ "enum": [
130
+ "local",
131
+ "delegated"
132
+ ]
133
+ },
134
+ "independent": {
135
+ "type": "boolean"
136
+ },
137
+ "read_only": {
138
+ "type": "boolean"
139
+ },
140
+ "timeout_seconds": {
141
+ "type": "integer",
142
+ "minimum": 1
143
+ },
144
+ "token_budget": {
145
+ "type": [
146
+ "integer",
147
+ "null"
148
+ ],
149
+ "minimum": 1
150
+ },
151
+ "metadata": {
152
+ "type": "object"
153
+ }
154
+ },
155
+ "additionalProperties": false
156
+ }
@@ -1 +1,60 @@
1
- {"$schema":"http://json-schema.org/draft-07/schema#","type":"object","required":["task_id","status","staging_artifacts","validated","validation_issues","metrics","summary"],"properties":{"task_id":{"type":"string","minLength":1},"status":{"enum":["completed","failed","blocked"]},"staging_artifacts":{"type":"array","items":{"type":"object","required":["artifact_type"],"properties":{"artifact_type":{"type":"string","minLength":1}},"additionalProperties":true}},"validated":{"type":"boolean"},"validation_issues":{"type":"array","items":{"type":"string"}},"metrics":{"type":"object"},"summary":{"type":"string"}},"additionalProperties":false}
1
+ {
2
+ "$schema": "http://json-schema.org/draft-07/schema#",
3
+ "title": "WorkerResult",
4
+ "description": "The lead process's verdict on one worker's submission (engine/worker_result.py). validate_worker_output() reconstructs this from untrusted worker output: any worker-supplied validated flag is ignored, canonical-state artifacts are rejected, and only results that pass here may enter a Judge context.",
5
+ "type": "object",
6
+ "required": [
7
+ "task_id",
8
+ "status",
9
+ "staging_artifacts",
10
+ "validated",
11
+ "validation_issues",
12
+ "metrics",
13
+ "summary"
14
+ ],
15
+ "properties": {
16
+ "task_id": {
17
+ "type": "string",
18
+ "minLength": 1
19
+ },
20
+ "status": {
21
+ "enum": [
22
+ "completed",
23
+ "failed",
24
+ "blocked"
25
+ ]
26
+ },
27
+ "staging_artifacts": {
28
+ "type": "array",
29
+ "items": {
30
+ "type": "object",
31
+ "required": [
32
+ "artifact_type"
33
+ ],
34
+ "properties": {
35
+ "artifact_type": {
36
+ "type": "string",
37
+ "minLength": 1
38
+ }
39
+ },
40
+ "additionalProperties": true
41
+ }
42
+ },
43
+ "validated": {
44
+ "type": "boolean"
45
+ },
46
+ "validation_issues": {
47
+ "type": "array",
48
+ "items": {
49
+ "type": "string"
50
+ }
51
+ },
52
+ "metrics": {
53
+ "type": "object"
54
+ },
55
+ "summary": {
56
+ "type": "string"
57
+ }
58
+ },
59
+ "additionalProperties": false
60
+ }
@@ -5,52 +5,188 @@
5
5
  "description": "Output of the Evidence Tribunal: what the evidence supports, cannot support, and the recommended action. Decision is one of ADOPT / PILOT / REJECT / INSUFFICIENT EVIDENCE (expressed here as adopt|pilot|reject|insufficient_evidence). confidence 由 scripts/compute_confidence.py 确定性计算并覆盖模型值;confidence_score 是规则化指数(0-1),不是概率。扩展字段一律放在 extensions 内。",
6
6
  "type": "object",
7
7
  "additionalProperties": false,
8
- "required": ["decision_question", "recommended_action", "confidence"],
8
+ "required": [
9
+ "decision_question",
10
+ "recommended_action",
11
+ "confidence"
12
+ ],
9
13
  "properties": {
10
- "decision_question": { "type": "string" },
11
- "target_population": { "type": "string" },
12
- "target_context": { "type": "string" },
13
- "supported_claims": { "type": "array", "items": { "type": "string" } },
14
- "uncertain_claims": { "type": "array", "items": { "type": "string" } },
15
- "contradicted_claims": { "type": "array", "items": { "type": "string" } },
16
- "reason_for_disagreement": { "type": "string" },
17
- "methodology_summary": { "type": "string" },
18
- "outcome_specific_findings": { "type": "object", "additionalProperties": true },
19
- "short_term_effect": { "type": ["string", "null"] },
20
- "long_term_effect": { "type": ["string", "null"] },
21
- "transfer_effect": { "type": ["string", "null"] },
22
- "risk_effect": { "type": ["string", "null"] },
23
- "applicability": { "type": "object", "additionalProperties": true },
24
- "confidence": { "type": "string", "enum": ["High", "Moderate", "Low", "Insufficient"] },
14
+ "decision_question": {
15
+ "type": "string"
16
+ },
17
+ "target_population": {
18
+ "type": "string"
19
+ },
20
+ "target_context": {
21
+ "type": "string"
22
+ },
23
+ "supported_claims": {
24
+ "type": "array",
25
+ "items": {
26
+ "type": "string"
27
+ }
28
+ },
29
+ "uncertain_claims": {
30
+ "type": "array",
31
+ "items": {
32
+ "type": "string"
33
+ }
34
+ },
35
+ "contradicted_claims": {
36
+ "type": "array",
37
+ "items": {
38
+ "type": "string"
39
+ }
40
+ },
41
+ "reason_for_disagreement": {
42
+ "type": "string"
43
+ },
44
+ "methodology_summary": {
45
+ "type": "string"
46
+ },
47
+ "outcome_specific_findings": {
48
+ "type": "object",
49
+ "additionalProperties": true
50
+ },
51
+ "short_term_effect": {
52
+ "type": [
53
+ "string",
54
+ "null"
55
+ ]
56
+ },
57
+ "long_term_effect": {
58
+ "type": [
59
+ "string",
60
+ "null"
61
+ ]
62
+ },
63
+ "transfer_effect": {
64
+ "type": [
65
+ "string",
66
+ "null"
67
+ ]
68
+ },
69
+ "risk_effect": {
70
+ "type": [
71
+ "string",
72
+ "null"
73
+ ]
74
+ },
75
+ "applicability": {
76
+ "type": "object",
77
+ "additionalProperties": true
78
+ },
79
+ "confidence": {
80
+ "type": "string",
81
+ "enum": [
82
+ "High",
83
+ "Moderate",
84
+ "Low",
85
+ "Insufficient"
86
+ ]
87
+ },
25
88
  "confidence_score": {
26
- "type": ["number", "null"],
89
+ "type": [
90
+ "number",
91
+ "null"
92
+ ],
27
93
  "minimum": 0,
28
94
  "maximum": 1,
29
95
  "description": "规则化置信度指数(0-1),由 compute_confidence.py 覆盖模型值。不是概率,禁止宣传为百分比。"
30
96
  },
31
- "confidence_policy_version": { "type": "string", "description": "确定性置信度策略版本号(如 2026-08-12.v1)。" },
32
- "independent_studies": { "type": ["integer", "null"], "minimum": 0, "description": "独立研究数(按 study_id/source_id 去重)。" },
33
- "independent_samples": { "type": ["integer", "null"], "minimum": 0, "description": "独立样本数(按 sample_id 去重)。" },
97
+ "confidence_policy_version": {
98
+ "type": "string",
99
+ "description": "确定性置信度策略版本号(如 2026-08-12.v1)。"
100
+ },
101
+ "independent_studies": {
102
+ "type": [
103
+ "integer",
104
+ "null"
105
+ ],
106
+ "minimum": 0,
107
+ "description": "独立研究数(按 study_id/source_id 去重)。"
108
+ },
109
+ "independent_samples": {
110
+ "type": [
111
+ "integer",
112
+ "null"
113
+ ],
114
+ "minimum": 0,
115
+ "description": "独立样本数(按 sample_id 去重)。"
116
+ },
34
117
  "confidence_breakdown": {
35
118
  "type": "object",
36
119
  "additionalProperties": true,
37
120
  "description": "Rule-based components: evidence_quality, consistency, directness, evidence_count, independent_studies, independent_samples, conflict_penalty, unsupported_penalty."
38
121
  },
39
122
  "raw_model_confidence": {
40
- "type": ["string", "null"],
123
+ "type": [
124
+ "string",
125
+ "null"
126
+ ],
41
127
  "description": "模型原始 confidence 输出(被确定性值覆盖前的值,仅供审计比对)。"
42
128
  },
43
- "raw_model_confidence_breakdown": { "type": "object", "additionalProperties": true },
44
- "what_can_be_claimed": { "type": "array", "items": { "type": "string" } },
45
- "what_cannot_be_claimed": { "type": "array", "items": { "type": "string" } },
46
- "missing_evidence": { "type": "array", "items": { "type": "string" } },
47
- "recommended_action": { "type": "string", "enum": ["adopt", "pilot", "reject", "insufficient_evidence"] },
48
- "decision_rationale": { "type": "string" },
49
- "exceeds_evidence_boundary": { "type": "array", "items": { "type": "string" }, "description": "Conclusions that currently go beyond the evidence boundary." },
129
+ "raw_model_confidence_breakdown": {
130
+ "type": "object",
131
+ "additionalProperties": true
132
+ },
133
+ "what_can_be_claimed": {
134
+ "type": "array",
135
+ "items": {
136
+ "type": "string"
137
+ }
138
+ },
139
+ "what_cannot_be_claimed": {
140
+ "type": "array",
141
+ "items": {
142
+ "type": "string"
143
+ }
144
+ },
145
+ "missing_evidence": {
146
+ "type": "array",
147
+ "items": {
148
+ "type": "string"
149
+ }
150
+ },
151
+ "recommended_action": {
152
+ "type": "string",
153
+ "enum": [
154
+ "adopt",
155
+ "pilot",
156
+ "reject",
157
+ "insufficient_evidence"
158
+ ]
159
+ },
160
+ "decision_rationale": {
161
+ "type": "string"
162
+ },
163
+ "exceeds_evidence_boundary": {
164
+ "type": "array",
165
+ "items": {
166
+ "type": "string"
167
+ },
168
+ "description": "Conclusions that currently go beyond the evidence boundary."
169
+ },
50
170
  "extensions": {
51
171
  "type": "object",
52
172
  "description": "结构化扩展字段的统一容器(P1-01)。未列入本 schema 的字段必须放在这里,禁止在顶层新增属性。",
53
173
  "additionalProperties": true
174
+ },
175
+ "strongest_support": {
176
+ "type": "string",
177
+ "description": "The single strongest conclusion the evidence supports, as a complete reader-facing sentence (<=60 chars zh / ~15 words en). Written by the adjudicator, not assembled by the renderer."
178
+ },
179
+ "key_uncertainty": {
180
+ "type": "string",
181
+ "description": "The decision-relevant uncertainty or counter-evidence, as a complete reader-facing sentence (<=70 chars zh / ~18 words en)."
182
+ },
183
+ "main_risk": {
184
+ "type": "string",
185
+ "description": "The principal risk of acting, as a complete reader-facing sentence (<=60 chars zh / ~15 words en)."
186
+ },
187
+ "next_action": {
188
+ "type": "string",
189
+ "description": "The recommended next step, as a complete reader-facing sentence (<=80 chars zh / ~20 words en)."
54
190
  }
55
191
  }
56
192
  }
@@ -17,9 +17,15 @@ schemas/v4/evidence-library.schema.json using the repo's zero-dependency
17
17
  validator (scripts/validate_schema.py).
18
18
 
19
19
  Direction semantics (adoption-relevant, conservative):
20
- support -> evidence favors adopting the intervention (=> pilot)
21
- contradict -> evidence opposes adopting the intervention (=> reject)
20
+ support -> evidence favors adopting the intervention
21
+ (offline screening cap => pilot; never adopt)
22
+ contradict -> evidence opposes adopting the intervention
23
+ (oppose-only => reject; mixed with support is an unresolved
24
+ conflict and must fall to INSUFFICIENT_EVIDENCE via
25
+ engine.decision_policy.decision_outcome)
22
26
  neutral -> inconclusive
27
+ The library never emits adopt: full ADOPT is decided only by
28
+ engine.decision_policy.decision_outcome (High + support + direct primary).
23
29
  For gold units the coarse rule is: if a question's expected decision range is
24
30
  purely reject-oriented ("reject" present and "pilot" absent), its
25
31
  key_claims/key_supporting_sources are harmful evidence => contradict, and its
@@ -266,10 +272,14 @@ def build(generated_at: str | None = None) -> tuple[dict[str, Any], int]:
266
272
  "key_claims/key_supporting_sources/known_contradictions/correct_outcome_types)"
267
273
  "+ 3 个示例工作流 evidence.jsonl(ai-coding-assistant / ai-tutor / ai-writing-assistant)"
268
274
  "抽取生成;按 (source_id, outcome_token, claim_text) 去重合并。"
269
- "direction 语义为采纳方向:support=支持采纳(初步裁决=>pilot),"
270
- "contradict=反对采纳(=>reject),neutral=中性;金标准条目按 expected_decision_range "
275
+ "direction 语义为采纳方向:support=支持采纳(离线初筛上限 => pilot,永不输出 adopt),"
276
+ "contradict=反对采纳(oppose-only 时 => reject;与 support 并存属未决冲突,应交 "
277
+ "engine.decision_policy.decision_outcome 判为 INSUFFICIENT_EVIDENCE),neutral=中性;"
278
+ "金标准条目按 expected_decision_range "
271
279
  "粗粒度映射方向(纯 reject 问题反向映射),conflict 与混合方向问题的单条断言方向可能不精确。"
272
- "仅用于离线初步裁决(preliminary,保守),从不直接给出 adopt。"
280
+ "仅用于离线初步裁决(preliminary,保守),从不直接给出 adopt;"
281
+ "完整 ADOPT 只能由 engine.decision_policy.decision_outcome 给出"
282
+ "(High + support + 主要结果 directness 2)。"
273
283
  ),
274
284
  }
275
285
  _validate(library)
@@ -40,7 +40,9 @@ def bake(examples: Path, *, force: bool = False) -> list[dict]:
40
40
  if not force and manifest.is_file():
41
41
  try:
42
42
  prior = json.loads(manifest.read_text(encoding='utf-8'))
43
- valid = prior.get('cache_key') == cache_key and all(
43
+ main = prior.get('main_report') or {}
44
+ main_ok = bool(main.get('file')) and (directory / main['file']).is_file() and hashlib.sha256((directory / main['file']).read_bytes()).hexdigest() == main.get('sha256')
45
+ valid = prior.get('cache_key') == cache_key and main_ok and all(
44
46
  (out_dir / record['file']).is_file() and hashlib.sha256((out_dir / record['file']).read_bytes()).hexdigest() == record['sha256']
45
47
  for record in prior.get('reports', [])) and len(prior.get('reports', [])) == len(THEMES)
46
48
  if valid:
@@ -64,8 +66,22 @@ def bake(examples: Path, *, force: bool = False) -> list[dict]:
64
66
  raise RuntimeError(f'{directory.name}/{theme}: renderer rejected input\n{completed.stdout}\n{completed.stderr}')
65
67
  os.replace(temporary, target)
66
68
  records.append({'theme': theme, 'file': target.name, 'sha256': hashlib.sha256(target.read_bytes()).hexdigest()})
69
+ # Also refresh the pack-root report. This used to write only the themed
70
+ # variants, so examples/*/EduEvidence_Report.html kept whatever bytes it
71
+ # was first rendered with - the packaged example shipped a report the
72
+ # current renderer would not produce.
73
+ from_default = next((r for r in records if r['theme'] == 'claude'), records[0])
74
+ main_target = directory / 'EduEvidence_Report.html'
75
+ main_temp = directory / '.EduEvidence_Report.pending.html'
76
+ main_temp.write_bytes((out_dir / from_default['file']).read_bytes())
77
+ os.replace(main_temp, main_target)
78
+ main_sha = hashlib.sha256(main_target.read_bytes()).hexdigest()
79
+
67
80
  value = {'schema_version': 1, 'project': directory.name, 'cache_key': cache_key,
68
- 'result_sha256': result_hash, 'renderer_sha256': engine_hash, 'reports': records}
81
+ 'result_sha256': result_hash, 'renderer_sha256': engine_hash,
82
+ 'main_report': {'file': main_target.name, 'sha256': main_sha,
83
+ 'theme': from_default['theme']},
84
+ 'reports': records}
69
85
  manifest.write_text(json.dumps(value, indent=2) + '\n', encoding='utf-8')
70
86
  reports.append(value)
71
87
  print(f'{directory.name}: {len(records)} verified report variants')
@@ -30,14 +30,14 @@ sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
30
30
  from evidence_semantics import effect_direction
31
31
  from engine.versions import ENGINE_VERSION
32
32
 
33
- OUTCOME_ORDER = [
34
- "knowledge_gain", "concept_understanding", "retention", "transfer",
35
- "independent_problem_solving", "completion_time", "accuracy",
36
- "code_quality", "assignment_score", "engagement", "motivation",
37
- "cognitive_load", "help_seeking", "metacognition", "ai_dependency",
38
- "over_reliance", "reduced_effort", "reduced_transfer",
39
- "academic_integrity_risk", "false_confidence",
40
- ]
33
+ def _outcome_order() -> list[str]:
34
+ """Registered outcome tokens in registry order (was a hard-coded list)."""
35
+ from engine.taxonomy import all_tokens_ordered
36
+
37
+ return list(all_tokens_ordered())
38
+
39
+
40
+ OUTCOME_ORDER = _outcome_order()
41
41
 
42
42
 
43
43
  def _load_json(path: Path) -> dict[str, Any] | None:
@@ -221,12 +221,74 @@ def build_claims(evidence: list[dict[str, Any]]) -> list[dict[str, Any]]:
221
221
  return list(claims.values())
222
222
 
223
223
 
224
+ def _applicability(pack_dir: Path, verdict: dict) -> dict:
225
+ """Stage-7 applicability assessment, falling back to the verdict boundary.
226
+
227
+ applicability.json is the dedicated deliverable of the Applicability stage.
228
+ It used to be written and then ignored, because the renderer only looked at
229
+ the verdict; this is where it re-enters the result.
230
+ """
231
+ import json as _json
232
+
233
+ path = pack_dir / "applicability.json"
234
+ if path.is_file():
235
+ try:
236
+ data = _json.loads(path.read_text(encoding="utf-8"))
237
+ except (OSError, _json.JSONDecodeError):
238
+ data = None
239
+ if isinstance(data, dict) and data and data.get("status") != "NOT_CAPTURED":
240
+ return data
241
+ value = verdict.get("applicability") if isinstance(verdict, dict) else None
242
+ return value if isinstance(value, dict) else {}
243
+
244
+ def _derive_study_audits(methodology: list[dict], evidence: list[dict]) -> list[dict]:
245
+ """Per-study audit rows derived from the audits and evidence present.
246
+
247
+ Each row names the study and reports the audit verdict that covers it,
248
+ so the per-study axis the schema advertises actually exists downstream.
249
+ Rows are only emitted for studies the audits or evidence actually name.
250
+ """
251
+ by_study: dict[str, dict] = {}
252
+ for audit in methodology:
253
+ if not isinstance(audit, dict):
254
+ continue
255
+ target = audit.get("target") or "overall"
256
+ if target == "overall":
257
+ # The aggregate audit is not a study row; label it as the
258
+ # body-of-evidence review so it cannot be mistaken for one.
259
+ target = "body_of_evidence"
260
+ entry = by_study.setdefault(target, {
261
+ "study_id": target,
262
+ "verdict": audit.get("verdict"),
263
+ "audit_items": audit.get("audit_items") or {},
264
+ "limitations": list(audit.get("limitations") or []),
265
+ "task_vs_learning_guard": audit.get("task_vs_learning_guard"),
266
+ })
267
+ entry.setdefault("evidence_ids", [])
268
+ known = {e.get("study_id") for e in evidence if e.get("study_id")}
269
+ for study_id in sorted(known):
270
+ by_study.setdefault(study_id, {
271
+ "study_id": study_id,
272
+ "verdict": None,
273
+ "audit_items": {},
274
+ "limitations": [],
275
+ "task_vs_learning_guard": None,
276
+ "evidence_ids": [e.get("evidence_id") for e in evidence
277
+ if e.get("study_id") == study_id],
278
+ })
279
+ return list(by_study.values())
280
+
281
+
224
282
  def build_result(pack_dir: Path, *, mode: str = "platform_native") -> dict[str, Any]:
225
283
  frame = _load_json(pack_dir / "frame.json") or {}
226
284
  evidence = _load_jsonl(pack_dir / "evidence.jsonl")
227
285
  # methodology.json is a single MethodologyAudit object (or a JSONL list)
228
286
  methodology_single = _load_json(pack_dir / "methodology.json")
229
287
  methodology = [methodology_single] if methodology_single else _load_jsonl(pack_dir / "methodology.jsonl")
288
+ # Per-study audits: the contract advertised them but nothing produced
289
+ # them, so a multi-study review silently shipped a single audit object.
290
+ # Derive the per-study axis from the audits actually present.
291
+ study_audits = _derive_study_audits(methodology, evidence)
230
292
  verdict = _load_json(pack_dir / "verdict.json") or {}
231
293
  intervention = _load_json(pack_dir / "intervention.json") or {}
232
294
  evaluation = _load_json(pack_dir / "evaluation.json") or {}
@@ -268,9 +330,12 @@ def build_result(pack_dir: Path, *, mode: str = "platform_native") -> dict[str,
268
330
  "sources": sources,
269
331
  "evidence": evidence,
270
332
  "methodology_reviews": methodology,
333
+ "study_audits": study_audits,
271
334
  "conflicts": [{"reason_for_disagreement": verdict.get("reason_for_disagreement", "")}]
272
335
  if verdict.get("reason_for_disagreement") else [],
273
- "applicability": verdict.get("applicability", {}),
336
+ # Prefer the dedicated stage-7 assessment; fall back to the verdict-embedded
337
+ # boundary when the run carries no separate applicability.json.
338
+ "applicability": _applicability(pack_dir, verdict),
274
339
  "intervention": intervention,
275
340
  "evaluation": evaluation,
276
341
  "benchmark": {},