eduevidence 6.0.0 → 6.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (267) hide show
  1. package/CHANGELOG.md +395 -0
  2. package/CONTRIBUTING.md +105 -0
  3. package/README.md +113 -49
  4. package/README.zh-CN.md +39 -12
  5. package/SKILL.md +15 -5
  6. package/assets/readme/landing-tour.gif +0 -0
  7. package/assets/readme/studio-tour.gif +0 -0
  8. package/benchmarks/evidence-library.json +277 -1
  9. package/bin/eduevidence.js +2 -1
  10. package/docs/architecture.md +325 -46
  11. package/docs/demo-workplace-ai.md +1 -1
  12. package/docs/install-guide.md +1 -1
  13. package/docs/j-ev-experimental.md +250 -0
  14. package/docs/orchestration-role-model.md +1 -1
  15. package/docs/release-closeout/README.md +1 -1
  16. package/docs/reproducibility.md +138 -0
  17. package/docs/sciverse-api.md +125 -0
  18. package/domains/_neutral/copy/few_shots.json +21 -0
  19. package/domains/_neutral/copy/framing_lexicon.json +19 -0
  20. package/domains/_neutral/copy/module_labels.json +5 -0
  21. package/domains/_neutral/copy/module_labels_footer.json +102 -0
  22. package/domains/_neutral/copy/module_labels_modules.json +204 -0
  23. package/domains/_neutral/copy/module_labels_nav.json +126 -0
  24. package/domains/_neutral/copy/module_labels_summary.json +98 -0
  25. package/domains/_neutral/copy/module_labels_tables.json +164 -0
  26. package/domains/_neutral/copy/module_labels_v2.json +90 -0
  27. package/domains/_neutral/copy/risk_constructs.json +20 -0
  28. package/domains/_neutral/copy/section_titles.json +66 -0
  29. package/domains/_neutral/copy/terminology.json +11 -0
  30. package/domains/check_copy_packs.py +103 -0
  31. package/domains/education/copy/few_shots.json +22 -0
  32. package/domains/education/copy/framing_enums.json +167 -0
  33. package/domains/education/copy/framing_lexicon.json +166 -0
  34. package/domains/education/copy/module_labels.json +169 -0
  35. package/domains/education/copy/risk_constructs.json +48 -0
  36. package/domains/education/copy/section_titles.json +186 -0
  37. package/domains/education/copy/terminology.json +70 -0
  38. package/domains/education/manifest.json +1 -1
  39. package/domains/education/outcome_taxonomy.json +2 -2
  40. package/domains/manifest.json +1 -1
  41. package/domains/policy/copy/few_shots.json +22 -0
  42. package/domains/policy/copy/framing_enums.json +94 -0
  43. package/domains/policy/copy/framing_lexicon.json +174 -0
  44. package/domains/policy/copy/module_labels.json +168 -0
  45. package/domains/policy/copy/risk_constructs.json +33 -0
  46. package/domains/policy/copy/section_titles.json +186 -0
  47. package/domains/policy/copy/terminology.json +64 -0
  48. package/eduevidence_cli.py +10 -0
  49. package/engine/capabilities.py +57 -5
  50. package/engine/decision_policy.py +167 -0
  51. package/engine/evidence_graph.py +14 -10
  52. package/engine/gaps.py +42 -22
  53. package/engine/ids.py +2 -0
  54. package/engine/library.py +6 -2
  55. package/engine/library_builtin.py +7 -4
  56. package/engine/living.py +34 -4
  57. package/engine/migration.py +88 -3
  58. package/engine/orchestration.py +5 -5
  59. package/engine/paths.py +2 -0
  60. package/engine/pilot.py +34 -32
  61. package/engine/taxonomy.py +211 -0
  62. package/engine/tribunal.py +49 -43
  63. package/engine/versions.py +1 -1
  64. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1361 -147
  65. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +3 -3
  66. package/examples/ai-coding-assistant-evidence/citation_check.json +1 -1
  67. package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
  68. package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
  69. package/examples/ai-coding-assistant-evidence/report_spec.json +23 -12
  70. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +448 -128
  71. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +448 -128
  72. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +448 -128
  73. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +448 -128
  74. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +448 -128
  75. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1360 -146
  76. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1360 -146
  77. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1360 -146
  78. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1360 -146
  79. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1360 -146
  80. package/examples/ai-coding-assistant-evidence/result.json +13 -9
  81. package/examples/ai-coding-assistant-evidence/result.zh.json +45 -41
  82. package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
  83. package/examples/ai-coding-assistant-evidence/verdict.json +6 -2
  84. package/examples/spaced-retrieval-practice/EduEvidence_Report.html +2728 -0
  85. package/examples/spaced-retrieval-practice/applicability.json +14 -0
  86. package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
  87. package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
  88. package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
  89. package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
  90. package/examples/spaced-retrieval-practice/frame.json +58 -0
  91. package/examples/spaced-retrieval-practice/gate_report.json +101 -0
  92. package/examples/spaced-retrieval-practice/methodology.json +78 -0
  93. package/examples/spaced-retrieval-practice/report.html +2522 -0
  94. package/examples/spaced-retrieval-practice/report_spec.json +212 -0
  95. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
  96. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
  97. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
  98. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
  99. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
  100. package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
  101. package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
  102. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
  103. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
  104. package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
  105. package/examples/spaced-retrieval-practice/result.json +942 -0
  106. package/examples/spaced-retrieval-practice/result.zh.json +942 -0
  107. package/examples/spaced-retrieval-practice/skeptic.json +70 -0
  108. package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
  109. package/examples/spaced-retrieval-practice/verdict.json +93 -0
  110. package/examples/workplace-ai-assistant/EduEvidence_Report.html +2814 -0
  111. package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
  112. package/examples/workplace-ai-assistant/claims.jsonl +4 -4
  113. package/examples/workplace-ai-assistant/evidence.jsonl +4 -4
  114. package/examples/workplace-ai-assistant/evidence_graph.json +15 -15
  115. package/examples/workplace-ai-assistant/final_verdict.json +78 -0
  116. package/examples/workplace-ai-assistant/gate_report.json +101 -0
  117. package/examples/workplace-ai-assistant/report_spec.json +209 -40
  118. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +449 -119
  119. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +449 -119
  120. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +449 -119
  121. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +449 -119
  122. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +449 -119
  123. package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
  124. package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
  125. package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
  126. package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
  127. package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
  128. package/examples/workplace-ai-assistant/result.json +82 -20
  129. package/examples/workplace-ai-assistant/result.zh.json +82 -20
  130. package/examples/workplace-ai-assistant/skeptic.json +72 -0
  131. package/examples/workplace-ai-assistant/verdict.json +36 -10
  132. package/integrations/agent_mcp.py +2 -2
  133. package/integrations/jev/__init__.py +115 -0
  134. package/integrations/jev/approval.py +212 -0
  135. package/integrations/jev/cli.py +84 -0
  136. package/integrations/jev/config.py +112 -0
  137. package/integrations/jev/gateway.py +128 -0
  138. package/integrations/jev/modes.py +38 -0
  139. package/integrations/jev/tools_classify.py +88 -0
  140. package/integrations/jev/tools_extract.py +111 -0
  141. package/integrations/jev/tools_rerank.py +71 -0
  142. package/integrations/jev/tools_screen.py +87 -0
  143. package/integrations/jev/tools_verify.py +95 -0
  144. package/integrations/jev_mcp.py +22 -0
  145. package/integrations/semantic_decide.py +286 -0
  146. package/integrations/semdecide_cli.py +55 -0
  147. package/package.json +19 -2
  148. package/pyproject.toml +4 -3
  149. package/references/report-copy-style.md +107 -0
  150. package/references/retrieval-compliance.md +75 -0
  151. package/references/retrieval-protocol.md +20 -0
  152. package/retrieval/audit.py +27 -3
  153. package/retrieval/fetch.py +96 -0
  154. package/retrieval/sciverse.py +398 -0
  155. package/retrieval/search.py +47 -7
  156. package/schemas/applicability.schema.json +94 -0
  157. package/schemas/chart-spec.schema.json +10 -3
  158. package/schemas/evidence.schema.json +316 -43
  159. package/schemas/fetch-result.schema.json +2 -1
  160. package/schemas/report-result.schema.json +3 -3
  161. package/schemas/report-spec.schema.json +98 -100
  162. package/schemas/skeptic.schema.json +86 -0
  163. package/schemas/source.schema.json +21 -2
  164. package/schemas/v2/decision-snapshot.schema.json +20 -9
  165. package/schemas/v2/finding.schema.json +5 -1
  166. package/schemas/v2/intake.schema.json +191 -0
  167. package/schemas/v2/methodology-audit.schema.json +5 -1
  168. package/schemas/v2/outcome.schema.json +28 -5
  169. package/schemas/v2/study.schema.json +5 -1
  170. package/schemas/vNext/autoevolve-session.schema.json +34 -1
  171. package/schemas/vNext/eval-snapshot.schema.json +77 -1
  172. package/schemas/vNext/execution-plan.schema.json +50 -1
  173. package/schemas/vNext/gap-priority.schema.json +54 -1
  174. package/schemas/vNext/negative-search-record.schema.json +68 -1
  175. package/schemas/vNext/research-iteration.schema.json +87 -1
  176. package/schemas/vNext/research-strategy.schema.json +62 -1
  177. package/schemas/vNext/skill-experiment.schema.json +90 -1
  178. package/schemas/vNext/task-spec.schema.json +156 -1
  179. package/schemas/vNext/worker-result.schema.json +60 -1
  180. package/schemas/verdict.schema.json +164 -28
  181. package/scripts/build_evidence_library.py +15 -5
  182. package/scripts/build_report_variants.py +18 -2
  183. package/scripts/build_result.py +74 -9
  184. package/scripts/check_package_parity.py +85 -0
  185. package/scripts/check_protocol_alignment.py +375 -0
  186. package/scripts/check_versioned_schemas.py +254 -0
  187. package/scripts/claim_audit.py +13 -8
  188. package/scripts/compute_confidence.py +10 -0
  189. package/scripts/dashboard_server.py +13 -2
  190. package/scripts/did_regression.py +12 -2
  191. package/scripts/evidence_score.py +5 -2
  192. package/scripts/intake/__init__.py +31 -0
  193. package/scripts/intake/__main__.py +18 -0
  194. package/scripts/intake/background.py +78 -0
  195. package/scripts/intake/browser.py +79 -0
  196. package/scripts/intake/cli.py +57 -0
  197. package/scripts/intake/constants.py +57 -0
  198. package/scripts/intake/depth.py +53 -0
  199. package/scripts/intake/enhancements.py +106 -0
  200. package/scripts/intake/hooks.py +90 -0
  201. package/scripts/intake/prefs.py +76 -0
  202. package/scripts/intake/prompts.py +85 -0
  203. package/scripts/intake/session.py +152 -0
  204. package/scripts/lint_file_layers.py +126 -0
  205. package/scripts/orchestrator.py +187 -40
  206. package/scripts/pre_verdict_gate.py +241 -29
  207. package/scripts/quickstart.py +18 -2
  208. package/scripts/run_workspace.py +7 -1
  209. package/scripts/skill_lint.py +11 -1
  210. package/scripts/skill_payload.py +6 -3
  211. package/scripts/test_adversarial_empirical.py +96 -25
  212. package/scripts/validate_schema.py +31 -1
  213. package/skill/agents/evaluation-designer.md +20 -4
  214. package/skill/agents/evidence-analyst.md +19 -3
  215. package/skill/agents/evidence-judge.md +98 -8
  216. package/skill/agents/evidence-retriever.md +20 -3
  217. package/skill/agents/intervention-designer.md +20 -4
  218. package/skill/agents/method-reviewer.md +18 -2
  219. package/skill/agents/{education-planner.md → research-planner.md} +19 -3
  220. package/skill/agents/skeptic.md +18 -2
  221. package/skill/roles/registry.yaml +11 -11
  222. package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
  223. package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
  224. package/skill/sub-skills/data-analysis/SKILL.md +34 -15
  225. package/skill/sub-skills/ethics-review/SKILL.md +33 -10
  226. package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
  227. package/skill/sub-skills/evidence-review/SKILL.md +31 -12
  228. package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
  229. package/skill/sub-skills/literature-review/SKILL.md +35 -14
  230. package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
  231. package/skill/sub-skills/report-generation/SKILL.md +28 -0
  232. package/skill/sub-skills/research-planning/SKILL.md +41 -14
  233. package/skill/sub-skills/study-design/SKILL.md +30 -9
  234. package/skill/task-briefs/adjudicate.md +32 -7
  235. package/skill/task-briefs/applicability.md +37 -2
  236. package/skill/task-briefs/audit.md +32 -7
  237. package/skill/task-briefs/challenge.md +34 -5
  238. package/skill/task-briefs/evaluate.md +30 -5
  239. package/skill/task-briefs/extract.md +31 -8
  240. package/skill/task-briefs/frame.md +39 -10
  241. package/skill/task-briefs/intervene.md +32 -6
  242. package/skill/task-briefs/present.md +32 -8
  243. package/skill/task-briefs/projection.md +36 -2
  244. package/skill/task-briefs/retrieve.md +36 -6
  245. package/skill/workflows/decision-and-pilot.md +76 -1
  246. package/skill/workflows/evaluate-and-update.md +83 -0
  247. package/skill/workflows/evidence-review.md +104 -0
  248. package/skill/workflows/experimental-jev.md +170 -0
  249. package/skill/workflows/intake.md +120 -0
  250. package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
  251. package/visualization/eduevidence-report/scripts/build_infographics.py +37 -15
  252. package/visualization/eduevidence-report/scripts/build_report.py +435 -575
  253. package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
  254. package/visualization/eduevidence-report/scripts/lieflat_engine.py +349 -38
  255. package/visualization/eduevidence-report/scripts/report_copy_pack.py +296 -0
  256. package/visualization/eduevidence-report/scripts/report_copy_policy_guard.py +47 -0
  257. package/visualization/eduevidence-report/scripts/zh_labels.py +141 -1
  258. package/web/architecture.html +14885 -0
  259. package/web/studio/assets/index-B8tkF44Q.css +1 -0
  260. package/web/studio/index.html +2 -2
  261. package/scripts/build_esl_artifacts.py +0 -1921
  262. package/scripts/build_killer_demo.py +0 -295
  263. package/scripts/enrich_projects_human_and_lieflat.py +0 -315
  264. package/scripts/generate_new_projects.py +0 -686
  265. package/scripts/sync_killer_demo_report.py +0 -270
  266. package/web/studio/assets/index-CzXocaGv.css +0 -1
  267. /package/web/studio/assets/{index-pa7jD7n4.js → index-CQ6Keoyc.js} +0 -0
@@ -1,13 +1,13 @@
1
1
  #!/usr/bin/env python3
2
2
  """pre_verdict_gate.py — Pre-Verdict Gate (Phase 15).
3
3
 
4
- An 11-item checklist that must pass (or be explicitly degraded) before a
4
+ A 12-item checklist that must pass (or be explicitly degraded) before a
5
5
  verdict may carry a high confidence label. The gate is deterministic and
6
6
  reads ONLY the run workspace — no model call, no network.
7
7
 
8
8
  Checklist:
9
9
 
10
- 1. research_frame_valid frame.json validates against education-frame.schema.json
10
+ 1. research_frame_valid frame.json validates against the run domain frame schema
11
11
  2. sources_valid sources.jsonl non-empty and schema-valid
12
12
  3. evidence_schema_valid evidence.jsonl non-empty and schema-valid
13
13
  4. source_dedupe no duplicate sources remain (dedupe applied)
@@ -58,7 +58,7 @@ for _p in (str(ROOT), str(ROOT / "scripts")):
58
58
 
59
59
  from validate_schema import SchemaError, Validator # noqa: E402
60
60
  from evidence_score import independent_samples, independent_studies # noqa: E402
61
- from evidence_semantics import claim_relation # noqa: E402
61
+ from evidence_semantics import claim_relation, decision_relation # noqa: E402
62
62
  from run_workspace import utc_now # noqa: E402
63
63
 
64
64
  GATE_VERSION = "2026-08-13.v1"
@@ -66,29 +66,52 @@ GATE_VERSION = "2026-08-13.v1"
66
66
  CONFIDENCE_RANK = {"Insufficient": 0, "Low": 1, "Moderate": 2, "High": 3}
67
67
 
68
68
  #: Advisory taxonomy for verdict outcome keys (shared with claim_audit).
69
- SUPPORTED_OUTCOMES = {
70
- "knowledge_gain", "concept_understanding", "retention", "transfer",
71
- "independent_problem_solving", "completion_time", "accuracy",
72
- "code_quality", "assignment_score", "engagement", "motivation",
73
- "cognitive_load", "help_seeking", "metacognition", "ai_dependency",
74
- "over_reliance", "reduced_effort", "reduced_transfer",
75
- "academic_integrity_risk", "false_confidence",
76
- }
69
+ def _supported_outcomes() -> set[str]:
70
+ """Every registered outcome token, read from the domain registry.
71
+
72
+ This was a hand-copied 20-token education list in three separate files;
73
+ it silently rejected policy tokens such as policy_effectiveness. The
74
+ registry (domains/<id>/outcome_taxonomy.json) is the single authority.
75
+ """
76
+ from engine.taxonomy import all_tokens_ordered
77
+
78
+ return set(all_tokens_ordered())
79
+
80
+
81
+ SUPPORTED_OUTCOMES = _supported_outcomes()
77
82
 
78
83
  _CLAIM_ID_RE = re.compile(r"\b[A-Z]{1,3}-\d{2,4}\b")
79
84
 
80
85
  _SCHEMA_CACHE: dict[str, dict[str, Any]] = {}
81
86
 
82
87
 
88
+ def _schema_path(name: str) -> Path:
89
+ """Resolve a schema name to a file.
90
+
91
+ Accepts a bare name in schemas/ as well as a repository-relative path,
92
+ because a domain may own its frame contract outside schemas/ (policy does:
93
+ domains/policy/frame.schema.json).
94
+ """
95
+ candidate = ROOT / "schemas" / name
96
+ if candidate.is_file():
97
+ return candidate
98
+ owned = ROOT / name
99
+ if owned.is_file():
100
+ return owned
101
+ return candidate # missing: the caller reports it
102
+
103
+
83
104
  def _schema(name: str) -> dict[str, Any]:
84
- if name not in _SCHEMA_CACHE:
85
- _SCHEMA_CACHE[name] = json.loads((ROOT / "schemas" / name).read_text(encoding="utf-8"))
86
- return _SCHEMA_CACHE[name]
105
+ path = _schema_path(name)
106
+ key = str(path)
107
+ if key not in _SCHEMA_CACHE:
108
+ _SCHEMA_CACHE[key] = json.loads(path.read_text(encoding="utf-8"))
109
+ return _SCHEMA_CACHE[key]
87
110
 
88
111
 
89
112
  def _validate_records(records: list[dict[str, Any]], schema_name: str, path: str) -> list[str]:
90
113
  schema = _schema(schema_name)
91
- validator = Validator(schema, base_dir=(ROOT / "schemas"))
114
+ validator = Validator(schema, base_dir=_schema_path(schema_name).parent)
92
115
  errors = []
93
116
  for idx, record in enumerate(records):
94
117
  try:
@@ -133,11 +156,43 @@ def _item_res(status: str, detail: str, *, blocks_high: bool | None = None) -> d
133
156
  return res
134
157
 
135
158
 
159
+ def workspace_domain(ws: Path) -> str:
160
+ """The domain this run/pack registered; defaults to education when absent.
161
+
162
+ A run workspace records it in run_manifest.json, but an example pack has
163
+ no manifest: reading only that file made a policy pack validate against the
164
+ education frame schema and fail item 1. Fall back to the same declared
165
+ places the read model uses, in the same order.
166
+ """
167
+ manifest = _load_ws_json(ws, "run_manifest.json")
168
+ if manifest.get("domain"):
169
+ return str(manifest["domain"])
170
+ frame = _load_ws_json(ws, "frame.json")
171
+ declared = (frame.get("extensions") or {}).get("domain")
172
+ if declared:
173
+ return str(declared)
174
+ result_meta = (_load_ws_json(ws, "result.json").get("meta") or {})
175
+ if result_meta.get("domain"):
176
+ return str(result_meta["domain"])
177
+ return "education"
178
+
179
+
180
+ def frame_schema_name(domain: str) -> str:
181
+ """Registered frame schema for a domain (repository-relative path)."""
182
+ from engine.evidencecore import load_domain
183
+
184
+ return str(load_domain(domain)["frame_schema"])
185
+
186
+
136
187
  def check_research_frame(ws: Path) -> dict[str, str]:
137
188
  frame = _load_ws_json(ws, "frame.json")
138
189
  if not frame:
139
190
  return _item_res("fail", "frame.json missing or empty (research question not framed)")
140
- errors = _validate_records([frame], "education-frame.schema.json", "frame")
191
+ try:
192
+ schema_name = frame_schema_name(workspace_domain(ws))
193
+ except KeyError as exc:
194
+ return _item_res("fail", f"run declares an unknown domain: {exc}")
195
+ errors = _validate_records([frame], schema_name, "frame")
141
196
  if errors:
142
197
  return _item_res("fail", f"frame.json schema invalid: {errors[0]}")
143
198
  return _item_res("pass", f"frame.json valid (question={frame.get('question', '')[:80]})")
@@ -182,11 +237,41 @@ def check_counter_evidence(ws: Path) -> dict[str, str]:
182
237
  return _item_res("fail", "skeptic.json missing or empty (counter-evidence search not performed)")
183
238
  if skeptic.get("search_performed") is not True:
184
239
  return _item_res("fail", "skeptic.json lacks search_performed=true")
185
- contradictions = skeptic.get("contradictions", []) or []
186
- null_results = skeptic.get("null_results", []) or []
187
- detail = f"search_performed=true; contradictions={len(contradictions)}, null_results={len(null_results)}"
188
- if skeptic.get("no_contradictory_evidence_found"):
189
- detail += "; no contradictory evidence found"
240
+
241
+ # The nine fixed checks are the contract (skill/task-briefs/challenge.md).
242
+ # A two-key shell used to pass this gate, which made the counter-evidence
243
+ # check cosmetic on every deterministic path.
244
+ required_checks = (
245
+ "1_null_result", "2_negative_result", "3_contradictory_evidence",
246
+ "4_alternative_explanation", "5_measurement_mismatch", "6_sampling_bias",
247
+ "7_novelty_effect", "8_ai_dependency", "9_scope_overreach",
248
+ )
249
+ findings = skeptic.get("skeptic_findings")
250
+ if not isinstance(findings, list) or not findings:
251
+ return _item_res(
252
+ "fail", "skeptic.json lacks skeptic_findings[] (nine fixed checks not run)")
253
+ present = {str(f.get("check")) for f in findings if isinstance(f, dict)}
254
+ missing = [c for c in required_checks if c not in present]
255
+ if missing:
256
+ return _item_res("fail", f"skeptic findings incomplete; missing {missing}")
257
+ for item in findings:
258
+ if not isinstance(item, dict) or item.get("status") not in ("found", "not_found"):
259
+ return _item_res(
260
+ "fail", f"skeptic finding {item.get('check')!r} lacks a found/not_found status")
261
+
262
+ found = bool(skeptic.get("contradictory_evidence_found"))
263
+ statement = skeptic.get("no_contradictory_evidence_statement")
264
+ if not found and not statement:
265
+ # Absence of counter-evidence must be asserted, not left implicit.
266
+ return _item_res(
267
+ "fail", "no contradictory evidence found but no statement recorded")
268
+ if found and statement:
269
+ return _item_res(
270
+ "fail", "contradictory_evidence_found=true together with a no-evidence statement")
271
+
272
+ found_checks = sum(1 for f in findings if f.get("status") == "found")
273
+ detail = (f"search_performed=true; 9/9 checks run; findings={found_checks}; "
274
+ f"contradictory_evidence_found={found}")
190
275
  return _item_res("pass", detail)
191
276
 
192
277
 
@@ -270,6 +355,105 @@ def check_claim_evidence(ws: Path) -> dict[str, str]:
270
355
  return _item_res("pass", "all verdict claims bind to existing evidence with consistent categories")
271
356
 
272
357
 
358
+ def _primary_evidence_summary(ws: Path) -> dict[str, Any]:
359
+ """Re-derive ADOPT eligibility from the pack's own evidence records.
360
+
361
+ The gate audits an artifact another party wrote, so it cannot trust the
362
+ verdict about itself: primary-result directness is read back from the
363
+ corpus (frame primary outcomes + per-record D5 Directness), using the same
364
+ domain registry and directness threshold the tribunal imports.
365
+ """
366
+ from engine.decision_policy import (
367
+ ADOPT_DIRECTNESS,
368
+ outcome_category,
369
+ primary_effect_categories,
370
+ )
371
+
372
+ domain = workspace_domain(ws)
373
+ try:
374
+ primary = primary_effect_categories(domain)
375
+ except (KeyError, ValueError) as exc:
376
+ return {"domain": domain, "primary": (), "direct": False, "error": str(exc)}
377
+
378
+ frame = _load_ws_json(ws, "frame.json")
379
+ declared_primary = [
380
+ str(token) for token in ((frame.get("outcomes") or {}).get("primary") or [])
381
+ ]
382
+ evidence = _load_ws_jsonl(ws, "evidence.jsonl")
383
+ direct_hits: list[str] = []
384
+ for ev in evidence:
385
+ token = str(ev.get("outcome_type") or "")
386
+ category = outcome_category(domain, token, primary)
387
+ if category is None or category not in primary:
388
+ continue
389
+ dims = ev.get("quality_dimensions") or {}
390
+ d5 = dims.get("D5_directness")
391
+ if isinstance(d5, bool) or not isinstance(d5, int):
392
+ continue
393
+ if d5 >= ADOPT_DIRECTNESS:
394
+ direct_hits.append(str(ev.get("evidence_id") or "?"))
395
+ return {
396
+ "domain": domain,
397
+ "primary": primary,
398
+ "declared_primary": declared_primary,
399
+ "direct": bool(direct_hits),
400
+ "direct_evidence_ids": direct_hits,
401
+ }
402
+
403
+
404
+ def check_decision_action(ws: Path) -> dict[str, str]:
405
+ """The stated action must be the action the evidence supports.
406
+
407
+ A verdict cannot hand itself ADOPT: the gate requires High confidence, a
408
+ supporting decisive relation, and primary-outcome evidence at directness 2.
409
+ Anything less is capped to pilot - the conservative bound - because the
410
+ underlying evidence may still justify a bounded trial.
411
+ """
412
+ verdict = _verdict_for_audit(ws)
413
+ if not verdict:
414
+ return _item_res("warn", "no verdict artifact yet; action not audited")
415
+ action = str(verdict.get("recommended_action") or "").lower()
416
+ if action != "adopt":
417
+ label = action or "unset"
418
+ return _item_res("pass", "action=" + label + " is within the conservative bound")
419
+
420
+ from engine.decision_policy import (
421
+ ADOPT_REQUIRED_LABEL,
422
+ decision_outcome,
423
+ )
424
+
425
+ evidence = _load_ws_jsonl(ws, "evidence.jsonl")
426
+ relations = [decision_relation(ev) for ev in evidence]
427
+ decisive = {str(index): rel for index, rel in enumerate(relations)
428
+ if rel in ("support_adoption", "oppose_adoption",
429
+ "conditional", "conflict", "mixed")}
430
+ label = str(verdict.get("confidence") or "")
431
+ summary = _primary_evidence_summary(ws)
432
+ outcome = decision_outcome(
433
+ confidence_label=label,
434
+ decisive_relations=decisive,
435
+ has_direct_primary_evidence=bool(summary.get("direct")),
436
+ )
437
+ expected = outcome["action"]
438
+ downgrade_reason = outcome.get("downgrade_reason")
439
+ if expected == "ADOPT":
440
+ detail = "ADOPT is supported: " + label + " confidence with direct primary-outcome evidence"
441
+ return _item_res("pass", detail)
442
+ reasons: list[str] = []
443
+ if label != ADOPT_REQUIRED_LABEL:
444
+ reasons.append("confidence=" + (label or "unset") + " (needs " + ADOPT_REQUIRED_LABEL + ")")
445
+ if "support_adoption" not in relations:
446
+ reasons.append("no decisive supporting evidence")
447
+ if not summary.get("direct"):
448
+ reasons.append("no primary-outcome evidence at directness 2")
449
+ if downgrade_reason:
450
+ reasons.append("downgrade_reason=" + downgrade_reason)
451
+ detail = ("recommended_action=adopt is not supported (" + "; ".join(reasons)
452
+ + "); the evidence bounds this decision to "
453
+ + str(expected).lower())
454
+ return _item_res("fail", detail)
455
+
456
+
273
457
  def check_outcome_mapping(ws: Path) -> dict[str, str]:
274
458
  verdict = _verdict_for_audit(ws)
275
459
  frame = _load_ws_json(ws, "frame.json")
@@ -282,18 +466,35 @@ def check_outcome_mapping(ws: Path) -> dict[str, str]:
282
466
  issues.append(f"unknown outcome key(s) in verdict: {', '.join(sorted(unknown))}")
283
467
 
284
468
  evidence_outcomes = {e.get("outcome_type") for e in _load_ws_jsonl(ws, "evidence.jsonl")}
469
+ frame_outcomes = frame.get("outcomes", {}) or {}
285
470
  declared = set()
286
471
  for group in ("primary", "secondary", "risk"):
287
- declared.update((frame.get("outcomes", {}) or {}).get(group, []) or [])
472
+ declared.update(frame_outcomes.get(group, []) or [])
288
473
  if declared:
289
474
  missing = sorted(d for d in declared if d and d not in evidence_outcomes)
290
475
  if missing:
291
- notes.append(f"frame-declared outcomes without evidence: {', '.join(missing)}")
476
+ # Missing evidence is not a zero effect, and a secondary or risk
477
+ # outcome the frame named but the corpus never measured does not
478
+ # contaminate the decision. Only a PRIMARY outcome with no
479
+ # evidence at all means the decision rests on the wrong construct,
480
+ # so only that case blocks High confidence.
481
+ primary_missing = sorted(
482
+ d for d in (frame_outcomes.get("primary", []) or [])
483
+ if d and d not in evidence_outcomes)
484
+ note = f"frame-declared outcomes without evidence: {', '.join(missing)}"
485
+ if primary_missing:
486
+ return _item_res(
487
+ "warn",
488
+ note + f"; primary outcomes unmeasured: {', '.join(primary_missing)}",
489
+ blocks_high=True)
490
+ notes.append(note)
491
+ else:
492
+ notes.append("all frame-declared outcomes are covered by evidence")
292
493
 
293
494
  if issues:
294
495
  return _item_res("fail", "; ".join(issues))
295
- if notes:
296
- return _item_res("warn", "outcome keys known; " + notes[0])
496
+ if notes and "without evidence" in notes[0]:
497
+ return _item_res("warn", "outcome keys known; " + notes[0], blocks_high=False)
297
498
  if not declared:
298
499
  return _item_res("warn", "outcome keys known; frame declares no outcomes to map")
299
500
  return _item_res("pass", f"outcome mapping complete ({len(evidence_outcomes)} outcome type(s) covered)")
@@ -375,18 +576,20 @@ GATE_ITEMS: list[dict[str, Any]] = [
375
576
  {"id": "claim_evidence_audit", "title": "Claim-Evidence Audit", "critical": True,
376
577
  "blocks_high": False, "check": check_claim_evidence},
377
578
  {"id": "outcome_mapping", "title": "Outcome mapping", "critical": False,
378
- "blocks_high": False, "check": check_outcome_mapping},
579
+ "blocks_high": True, "check": check_outcome_mapping},
379
580
  {"id": "scope_calibration", "title": "Scope calibration", "critical": False,
380
- "blocks_high": False, "check": check_scope_calibration},
581
+ "blocks_high": True, "check": check_scope_calibration},
381
582
  {"id": "independent_study_count", "title": "Independent study-sample count", "critical": True,
382
583
  "blocks_high": True, "check": check_study_count},
383
584
  {"id": "deterministic_confidence", "title": "Deterministic confidence", "critical": True,
384
585
  "blocks_high": True, "check": None},
586
+ {"id": "decision_action_consistency", "title": "Decision action consistency",
587
+ "critical": True, "blocks_high": False, "check": check_decision_action},
385
588
  ]
386
589
 
387
590
 
388
591
  def evaluate_workspace(workspace: Path, *, require_final: bool = True) -> dict[str, Any]:
389
- """Run the 11-item gate over a run workspace. Pure, deterministic, read-only."""
592
+ """Run the 12-item gate over a run workspace. Pure, deterministic, read-only."""
390
593
  ws = Path(workspace)
391
594
  items: dict[str, dict[str, Any]] = {}
392
595
  for spec in GATE_ITEMS:
@@ -437,6 +640,7 @@ def apply_enforcement(verdict: dict[str, Any], gate: dict[str, Any]) -> dict[str
437
640
 
438
641
  - gate failed -> confidence at most Low; adopt downgraded to pilot
439
642
  - gate passed but High blocked -> confidence at most Moderate
643
+ - stated adopt the evidence does not support -> downgraded to pilot
440
644
  Always records the enforcement inside verdict.extensions.gate_enforcement.
441
645
  """
442
646
  import copy
@@ -446,8 +650,15 @@ def apply_enforcement(verdict: dict[str, Any], gate: dict[str, Any]) -> dict[str
446
650
  current = out.get("confidence", "Insufficient")
447
651
  if CONFIDENCE_RANK.get(current, 0) > CONFIDENCE_RANK.get(cap, 0):
448
652
  out["confidence"] = cap
653
+ downgrade_reason = None
449
654
  if not gate.get("passed", False) and out.get("recommended_action") == "adopt":
450
655
  out["recommended_action"] = "pilot"
656
+ downgrade_reason = "gate_critical"
657
+ action_item = (gate.get("items") or {}).get("decision_action_consistency") or {}
658
+ if action_item.get("status") == "fail" and out.get("recommended_action") == "adopt":
659
+ out["recommended_action"] = "pilot"
660
+ if downgrade_reason is None:
661
+ downgrade_reason = "missing_direct_primary"
451
662
  extensions = out.setdefault("extensions", {})
452
663
  if not isinstance(extensions, dict):
453
664
  extensions = {}
@@ -461,6 +672,7 @@ def apply_enforcement(verdict: dict[str, Any], gate: dict[str, Any]) -> dict[str
461
672
  "max_confidence": cap,
462
673
  "confidence_before": current,
463
674
  "action_before": verdict.get("recommended_action"),
675
+ "downgrade_reason": downgrade_reason,
464
676
  })
465
677
  return out
466
678
 
@@ -469,7 +681,7 @@ def apply_enforcement(verdict: dict[str, Any], gate: dict[str, Any]) -> dict[str
469
681
 
470
682
 
471
683
  def main(argv: list[str] | None = None) -> int:
472
- parser = argparse.ArgumentParser(description="EduEvidence Pre-Verdict Gate (11-item checklist)")
684
+ parser = argparse.ArgumentParser(description="EduEvidence Pre-Verdict Gate (12-item checklist)")
473
685
  parser.add_argument("--workspace", required=True, help="run workspace directory (runs/<run_id>)")
474
686
  parser.add_argument("--require-final", action="store_true",
475
687
  help="item 11 fails when final_verdict.json is missing (default: warn)")
@@ -40,8 +40,23 @@ STAGE_BRIEFS = {
40
40
  }
41
41
 
42
42
 
43
+ def runs_root() -> Path:
44
+ """Runs directory, honouring EDUEVIDENCE_RUNS_DIR like the orchestrator does.
45
+
46
+ Without this the quickstart wrote into the repository even when the user had
47
+ pointed the runs directory elsewhere.
48
+ """
49
+ import os
50
+
51
+ return Path(os.environ.get("EDUEVIDENCE_RUNS_DIR") or (ROOT / "runs"))
52
+
53
+
43
54
  def newest_run_dir() -> Path:
44
- runs_dir = ROOT / "runs"
55
+ runs_dir = runs_root()
56
+ if not runs_dir.is_dir():
57
+ raise SystemExit(
58
+ f"no runs directory at {runs_dir}; run `eduevidence run --question ...` first "
59
+ "(or set EDUEVIDENCE_RUNS_DIR)")
45
60
  candidates = sorted((p for p in runs_dir.iterdir() if p.is_dir()),
46
61
  key=lambda p: p.stat().st_mtime, reverse=True)
47
62
  if not candidates:
@@ -86,7 +101,8 @@ def build_next_steps(run_dir: Path, question: str, depth: str) -> str:
86
101
  "",
87
102
  "## 可信度自检",
88
103
  "",
89
- "- 引用逐条核验报告:`benchmarks/doi-audit/report.md` 与包内 `citation_check.md`",
104
+ "- 引用逐条核验报告:包内 `citation_check.md`"
105
+ "(源码仓库另见 `benchmarks/doi-audit/report.md`)",
90
106
  "- 报告头徽章标注 data_origin;synthetic 演示不得当作实证引用",
91
107
  ""]
92
108
  return "\n".join(lines)
@@ -179,14 +179,20 @@ def build_manifest(
179
179
  scp_available: bool | None = None,
180
180
  root: Path | None = None,
181
181
  started_at: str | None = None,
182
+ domain: str = "education",
182
183
  ) -> dict[str, Any]:
183
- """Phase 13 run manifest with every contract field."""
184
+ """Phase 13 run manifest with every contract field.
185
+
186
+ ``domain`` selects which registered frame contract this run must satisfy;
187
+ it defaults to education so existing callers and manifests stay valid.
188
+ """
184
189
  return {
185
190
  "run_id": run_id,
186
191
  "skill_version": SKILL_VERSION,
187
192
  "git_commit": git_commit(root),
188
193
  "started_at": started_at or utc_now(),
189
194
  "question": question,
195
+ "domain": domain,
190
196
  "execution_mode": execution_mode,
191
197
  "scp_available": detect_scp() if scp_available is None else scp_available,
192
198
  "agent_mcp_available": agent_mcp_available,
@@ -13,7 +13,6 @@ Usage:
13
13
  """
14
14
  from __future__ import annotations
15
15
 
16
- import os
17
16
  import re
18
17
  import sys
19
18
  from pathlib import Path
@@ -130,6 +129,17 @@ def lint_skill() -> list[str]:
130
129
  if re.search(r"theme-btn|theme-switcher|data-theme-target", ltext):
131
130
  errors.append("scripts/render_report_html.py still contains runtime theme switcher")
132
131
 
132
+ # 10. File-layer hard gate (also runnable alone: python3 scripts/lint_file_layers.py)
133
+ try:
134
+ sys.path.insert(0, str(ROOT / "scripts"))
135
+ from lint_file_layers import check_file_layers # noqa: PLC0415
136
+ layer_errors, layer_warnings = check_file_layers()
137
+ errors.extend(layer_errors)
138
+ for w in layer_warnings:
139
+ print(f" ! {w}")
140
+ except Exception as exc: # pragma: no cover - lint must still run
141
+ errors.append(f"file-layer gate failed to load: {exc}")
142
+
133
143
  return errors
134
144
 
135
145
 
@@ -7,12 +7,14 @@ import sys
7
7
 
8
8
  TREES = (
9
9
  "agents", "engine", "domains", "skill", "references", "schemas", "scripts",
10
+ "integrations", "visualization/eduevidence-report",
10
11
  "retrieval", "integrations", "visualization/eduevidence-report", "web/studio",
11
12
  "assets/readme",
12
13
  )
13
14
  FILES = (
14
15
  "SKILL.md", "eduevidence_cli.py", "install.sh", "pyproject.toml", "setup.py",
15
16
  "LICENSE", "CHANGELOG.md", "README.md", "README.zh-CN.md", "web/index.html",
17
+ "CONTRIBUTING.md", "web/architecture.html",
16
18
  "benchmarks/evidence-library.json", "benchmarks/partitions.json",
17
19
  "benchmarks/adversarial/cases.jsonl",
18
20
  )
@@ -21,11 +23,13 @@ DOCS = (
21
23
  "reproducibility.md", "release-contract.md", "autoresearch-evolution-plan.md",
22
24
  "orchestration-role-model.md", "autoresearch-implementation-status.md",
23
25
  "research-studio-guide.zh-CN.md", "demo-workplace-ai.md",
26
+ "sciverse-api.md", "j-ev-experimental.md",
24
27
  "release-closeout/README.md", "release-closeout/issues.md",
25
28
  "release-closeout/frontend-acceptance.md", "release-closeout/verification.md",
26
29
  )
27
30
  EXAMPLES = (
28
- "ai-coding-assistant-evidence", "workplace-ai-assistant",
31
+ "ai-coding-assistant-evidence", "spaced-retrieval-practice",
32
+ "workplace-ai-assistant",
29
33
  )
30
34
  EXAMPLE_FILES = (
31
35
  "result.json", "result.zh.json", "evidence_graph.json", "report_spec.json",
@@ -34,8 +38,7 @@ EXAMPLE_FILES = (
34
38
  )
35
39
  RETIRED_DEMO_SCRIPTS = {
36
40
  "scripts/build_esl_artifacts.py", "scripts/generate_new_projects.py",
37
- "scripts/enrich_projects_human_and_lieflat.py", "scripts/build_killer_demo.py",
38
- "scripts/sync_killer_demo_report.py",
41
+ "scripts/enrich_projects_human_and_lieflat.py",
39
42
  }
40
43
 
41
44