eduevidence 6.0.0 → 6.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (267) hide show
  1. package/CHANGELOG.md +395 -0
  2. package/CONTRIBUTING.md +105 -0
  3. package/README.md +113 -49
  4. package/README.zh-CN.md +39 -12
  5. package/SKILL.md +15 -5
  6. package/assets/readme/landing-tour.gif +0 -0
  7. package/assets/readme/studio-tour.gif +0 -0
  8. package/benchmarks/evidence-library.json +277 -1
  9. package/bin/eduevidence.js +2 -1
  10. package/docs/architecture.md +325 -46
  11. package/docs/demo-workplace-ai.md +1 -1
  12. package/docs/install-guide.md +1 -1
  13. package/docs/j-ev-experimental.md +250 -0
  14. package/docs/orchestration-role-model.md +1 -1
  15. package/docs/release-closeout/README.md +1 -1
  16. package/docs/reproducibility.md +138 -0
  17. package/docs/sciverse-api.md +125 -0
  18. package/domains/_neutral/copy/few_shots.json +21 -0
  19. package/domains/_neutral/copy/framing_lexicon.json +19 -0
  20. package/domains/_neutral/copy/module_labels.json +5 -0
  21. package/domains/_neutral/copy/module_labels_footer.json +102 -0
  22. package/domains/_neutral/copy/module_labels_modules.json +204 -0
  23. package/domains/_neutral/copy/module_labels_nav.json +126 -0
  24. package/domains/_neutral/copy/module_labels_summary.json +98 -0
  25. package/domains/_neutral/copy/module_labels_tables.json +164 -0
  26. package/domains/_neutral/copy/module_labels_v2.json +90 -0
  27. package/domains/_neutral/copy/risk_constructs.json +20 -0
  28. package/domains/_neutral/copy/section_titles.json +66 -0
  29. package/domains/_neutral/copy/terminology.json +11 -0
  30. package/domains/check_copy_packs.py +103 -0
  31. package/domains/education/copy/few_shots.json +22 -0
  32. package/domains/education/copy/framing_enums.json +167 -0
  33. package/domains/education/copy/framing_lexicon.json +166 -0
  34. package/domains/education/copy/module_labels.json +169 -0
  35. package/domains/education/copy/risk_constructs.json +48 -0
  36. package/domains/education/copy/section_titles.json +186 -0
  37. package/domains/education/copy/terminology.json +70 -0
  38. package/domains/education/manifest.json +1 -1
  39. package/domains/education/outcome_taxonomy.json +2 -2
  40. package/domains/manifest.json +1 -1
  41. package/domains/policy/copy/few_shots.json +22 -0
  42. package/domains/policy/copy/framing_enums.json +94 -0
  43. package/domains/policy/copy/framing_lexicon.json +174 -0
  44. package/domains/policy/copy/module_labels.json +168 -0
  45. package/domains/policy/copy/risk_constructs.json +33 -0
  46. package/domains/policy/copy/section_titles.json +186 -0
  47. package/domains/policy/copy/terminology.json +64 -0
  48. package/eduevidence_cli.py +10 -0
  49. package/engine/capabilities.py +57 -5
  50. package/engine/decision_policy.py +167 -0
  51. package/engine/evidence_graph.py +14 -10
  52. package/engine/gaps.py +42 -22
  53. package/engine/ids.py +2 -0
  54. package/engine/library.py +6 -2
  55. package/engine/library_builtin.py +7 -4
  56. package/engine/living.py +34 -4
  57. package/engine/migration.py +88 -3
  58. package/engine/orchestration.py +5 -5
  59. package/engine/paths.py +2 -0
  60. package/engine/pilot.py +34 -32
  61. package/engine/taxonomy.py +211 -0
  62. package/engine/tribunal.py +49 -43
  63. package/engine/versions.py +1 -1
  64. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1361 -147
  65. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +3 -3
  66. package/examples/ai-coding-assistant-evidence/citation_check.json +1 -1
  67. package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
  68. package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
  69. package/examples/ai-coding-assistant-evidence/report_spec.json +23 -12
  70. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +448 -128
  71. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +448 -128
  72. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +448 -128
  73. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +448 -128
  74. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +448 -128
  75. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1360 -146
  76. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1360 -146
  77. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1360 -146
  78. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1360 -146
  79. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1360 -146
  80. package/examples/ai-coding-assistant-evidence/result.json +13 -9
  81. package/examples/ai-coding-assistant-evidence/result.zh.json +45 -41
  82. package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
  83. package/examples/ai-coding-assistant-evidence/verdict.json +6 -2
  84. package/examples/spaced-retrieval-practice/EduEvidence_Report.html +2728 -0
  85. package/examples/spaced-retrieval-practice/applicability.json +14 -0
  86. package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
  87. package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
  88. package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
  89. package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
  90. package/examples/spaced-retrieval-practice/frame.json +58 -0
  91. package/examples/spaced-retrieval-practice/gate_report.json +101 -0
  92. package/examples/spaced-retrieval-practice/methodology.json +78 -0
  93. package/examples/spaced-retrieval-practice/report.html +2522 -0
  94. package/examples/spaced-retrieval-practice/report_spec.json +212 -0
  95. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
  96. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
  97. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
  98. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
  99. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
  100. package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
  101. package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
  102. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
  103. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
  104. package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
  105. package/examples/spaced-retrieval-practice/result.json +942 -0
  106. package/examples/spaced-retrieval-practice/result.zh.json +942 -0
  107. package/examples/spaced-retrieval-practice/skeptic.json +70 -0
  108. package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
  109. package/examples/spaced-retrieval-practice/verdict.json +93 -0
  110. package/examples/workplace-ai-assistant/EduEvidence_Report.html +2814 -0
  111. package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
  112. package/examples/workplace-ai-assistant/claims.jsonl +4 -4
  113. package/examples/workplace-ai-assistant/evidence.jsonl +4 -4
  114. package/examples/workplace-ai-assistant/evidence_graph.json +15 -15
  115. package/examples/workplace-ai-assistant/final_verdict.json +78 -0
  116. package/examples/workplace-ai-assistant/gate_report.json +101 -0
  117. package/examples/workplace-ai-assistant/report_spec.json +209 -40
  118. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +449 -119
  119. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +449 -119
  120. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +449 -119
  121. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +449 -119
  122. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +449 -119
  123. package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
  124. package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
  125. package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
  126. package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
  127. package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
  128. package/examples/workplace-ai-assistant/result.json +82 -20
  129. package/examples/workplace-ai-assistant/result.zh.json +82 -20
  130. package/examples/workplace-ai-assistant/skeptic.json +72 -0
  131. package/examples/workplace-ai-assistant/verdict.json +36 -10
  132. package/integrations/agent_mcp.py +2 -2
  133. package/integrations/jev/__init__.py +115 -0
  134. package/integrations/jev/approval.py +212 -0
  135. package/integrations/jev/cli.py +84 -0
  136. package/integrations/jev/config.py +112 -0
  137. package/integrations/jev/gateway.py +128 -0
  138. package/integrations/jev/modes.py +38 -0
  139. package/integrations/jev/tools_classify.py +88 -0
  140. package/integrations/jev/tools_extract.py +111 -0
  141. package/integrations/jev/tools_rerank.py +71 -0
  142. package/integrations/jev/tools_screen.py +87 -0
  143. package/integrations/jev/tools_verify.py +95 -0
  144. package/integrations/jev_mcp.py +22 -0
  145. package/integrations/semantic_decide.py +286 -0
  146. package/integrations/semdecide_cli.py +55 -0
  147. package/package.json +19 -2
  148. package/pyproject.toml +4 -3
  149. package/references/report-copy-style.md +107 -0
  150. package/references/retrieval-compliance.md +75 -0
  151. package/references/retrieval-protocol.md +20 -0
  152. package/retrieval/audit.py +27 -3
  153. package/retrieval/fetch.py +96 -0
  154. package/retrieval/sciverse.py +398 -0
  155. package/retrieval/search.py +47 -7
  156. package/schemas/applicability.schema.json +94 -0
  157. package/schemas/chart-spec.schema.json +10 -3
  158. package/schemas/evidence.schema.json +316 -43
  159. package/schemas/fetch-result.schema.json +2 -1
  160. package/schemas/report-result.schema.json +3 -3
  161. package/schemas/report-spec.schema.json +98 -100
  162. package/schemas/skeptic.schema.json +86 -0
  163. package/schemas/source.schema.json +21 -2
  164. package/schemas/v2/decision-snapshot.schema.json +20 -9
  165. package/schemas/v2/finding.schema.json +5 -1
  166. package/schemas/v2/intake.schema.json +191 -0
  167. package/schemas/v2/methodology-audit.schema.json +5 -1
  168. package/schemas/v2/outcome.schema.json +28 -5
  169. package/schemas/v2/study.schema.json +5 -1
  170. package/schemas/vNext/autoevolve-session.schema.json +34 -1
  171. package/schemas/vNext/eval-snapshot.schema.json +77 -1
  172. package/schemas/vNext/execution-plan.schema.json +50 -1
  173. package/schemas/vNext/gap-priority.schema.json +54 -1
  174. package/schemas/vNext/negative-search-record.schema.json +68 -1
  175. package/schemas/vNext/research-iteration.schema.json +87 -1
  176. package/schemas/vNext/research-strategy.schema.json +62 -1
  177. package/schemas/vNext/skill-experiment.schema.json +90 -1
  178. package/schemas/vNext/task-spec.schema.json +156 -1
  179. package/schemas/vNext/worker-result.schema.json +60 -1
  180. package/schemas/verdict.schema.json +164 -28
  181. package/scripts/build_evidence_library.py +15 -5
  182. package/scripts/build_report_variants.py +18 -2
  183. package/scripts/build_result.py +74 -9
  184. package/scripts/check_package_parity.py +85 -0
  185. package/scripts/check_protocol_alignment.py +375 -0
  186. package/scripts/check_versioned_schemas.py +254 -0
  187. package/scripts/claim_audit.py +13 -8
  188. package/scripts/compute_confidence.py +10 -0
  189. package/scripts/dashboard_server.py +13 -2
  190. package/scripts/did_regression.py +12 -2
  191. package/scripts/evidence_score.py +5 -2
  192. package/scripts/intake/__init__.py +31 -0
  193. package/scripts/intake/__main__.py +18 -0
  194. package/scripts/intake/background.py +78 -0
  195. package/scripts/intake/browser.py +79 -0
  196. package/scripts/intake/cli.py +57 -0
  197. package/scripts/intake/constants.py +57 -0
  198. package/scripts/intake/depth.py +53 -0
  199. package/scripts/intake/enhancements.py +106 -0
  200. package/scripts/intake/hooks.py +90 -0
  201. package/scripts/intake/prefs.py +76 -0
  202. package/scripts/intake/prompts.py +85 -0
  203. package/scripts/intake/session.py +152 -0
  204. package/scripts/lint_file_layers.py +126 -0
  205. package/scripts/orchestrator.py +187 -40
  206. package/scripts/pre_verdict_gate.py +241 -29
  207. package/scripts/quickstart.py +18 -2
  208. package/scripts/run_workspace.py +7 -1
  209. package/scripts/skill_lint.py +11 -1
  210. package/scripts/skill_payload.py +6 -3
  211. package/scripts/test_adversarial_empirical.py +96 -25
  212. package/scripts/validate_schema.py +31 -1
  213. package/skill/agents/evaluation-designer.md +20 -4
  214. package/skill/agents/evidence-analyst.md +19 -3
  215. package/skill/agents/evidence-judge.md +98 -8
  216. package/skill/agents/evidence-retriever.md +20 -3
  217. package/skill/agents/intervention-designer.md +20 -4
  218. package/skill/agents/method-reviewer.md +18 -2
  219. package/skill/agents/{education-planner.md → research-planner.md} +19 -3
  220. package/skill/agents/skeptic.md +18 -2
  221. package/skill/roles/registry.yaml +11 -11
  222. package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
  223. package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
  224. package/skill/sub-skills/data-analysis/SKILL.md +34 -15
  225. package/skill/sub-skills/ethics-review/SKILL.md +33 -10
  226. package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
  227. package/skill/sub-skills/evidence-review/SKILL.md +31 -12
  228. package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
  229. package/skill/sub-skills/literature-review/SKILL.md +35 -14
  230. package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
  231. package/skill/sub-skills/report-generation/SKILL.md +28 -0
  232. package/skill/sub-skills/research-planning/SKILL.md +41 -14
  233. package/skill/sub-skills/study-design/SKILL.md +30 -9
  234. package/skill/task-briefs/adjudicate.md +32 -7
  235. package/skill/task-briefs/applicability.md +37 -2
  236. package/skill/task-briefs/audit.md +32 -7
  237. package/skill/task-briefs/challenge.md +34 -5
  238. package/skill/task-briefs/evaluate.md +30 -5
  239. package/skill/task-briefs/extract.md +31 -8
  240. package/skill/task-briefs/frame.md +39 -10
  241. package/skill/task-briefs/intervene.md +32 -6
  242. package/skill/task-briefs/present.md +32 -8
  243. package/skill/task-briefs/projection.md +36 -2
  244. package/skill/task-briefs/retrieve.md +36 -6
  245. package/skill/workflows/decision-and-pilot.md +76 -1
  246. package/skill/workflows/evaluate-and-update.md +83 -0
  247. package/skill/workflows/evidence-review.md +104 -0
  248. package/skill/workflows/experimental-jev.md +170 -0
  249. package/skill/workflows/intake.md +120 -0
  250. package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
  251. package/visualization/eduevidence-report/scripts/build_infographics.py +37 -15
  252. package/visualization/eduevidence-report/scripts/build_report.py +435 -575
  253. package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
  254. package/visualization/eduevidence-report/scripts/lieflat_engine.py +349 -38
  255. package/visualization/eduevidence-report/scripts/report_copy_pack.py +296 -0
  256. package/visualization/eduevidence-report/scripts/report_copy_policy_guard.py +47 -0
  257. package/visualization/eduevidence-report/scripts/zh_labels.py +141 -1
  258. package/web/architecture.html +14885 -0
  259. package/web/studio/assets/index-B8tkF44Q.css +1 -0
  260. package/web/studio/index.html +2 -2
  261. package/scripts/build_esl_artifacts.py +0 -1921
  262. package/scripts/build_killer_demo.py +0 -295
  263. package/scripts/enrich_projects_human_and_lieflat.py +0 -315
  264. package/scripts/generate_new_projects.py +0 -686
  265. package/scripts/sync_killer_demo_report.py +0 -270
  266. package/web/studio/assets/index-CzXocaGv.css +0 -1
  267. /package/web/studio/assets/{index-pa7jD7n4.js → index-CQ6Keoyc.js} +0 -0
@@ -0,0 +1,254 @@
1
+ #!/usr/bin/env python3
2
+ """check_versioned_schemas.py - data-level validation for v3 / v4 / vNext.
3
+
4
+ CI validated this contract family by parsing the JSON only, so a schema could
5
+ drift arbitrarily (renamed required fields, changed enums, wrong types) and
6
+ stay green. Each family is exercised here by building a record from its own
7
+ dataclass or builder and validating that record against its schema, which is
8
+ the check that would have caught a real mismatch.
9
+
10
+ Stdlib only; exit 1 on any invalid record.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import json
15
+ import sys
16
+ from pathlib import Path
17
+
18
+ ROOT = Path(__file__).resolve().parent.parent
19
+ sys.path.insert(0, str(ROOT))
20
+ sys.path.insert(0, str(ROOT / 'scripts'))
21
+
22
+
23
+ def _validate(record, schema_rel, label):
24
+ from validate_schema import SchemaError, Validator
25
+
26
+ path = ROOT / schema_rel
27
+ schema = json.loads(path.read_text(encoding='utf-8'))
28
+ validator = Validator(schema, base_dir=path.parent)
29
+ try:
30
+ validator.validate(record, schema, label)
31
+ except SchemaError as exc:
32
+ return f'{label}: {exc}'
33
+ return None
34
+
35
+
36
+ def main() -> int:
37
+ problems = []
38
+ checked = 0
39
+
40
+ # --- vNext: records built from their own dataclasses -------------------
41
+ try:
42
+ from engine.autoresearch.contracts import NegativeSearchRecord
43
+
44
+ record = NegativeSearchRecord(
45
+ negative_search_id='NSR-1', research_iteration_id='RI-1',
46
+ gap_id='GAP-1', queries=('a', 'b'), providers=('openalex',),
47
+ candidate_count=0, fetched_count=0, eligible_count=0)
48
+ payload = record.__dict__ if hasattr(record, '__dict__') else record
49
+ # JSON round-trip puts tuples back into their serialized form, which is
50
+ # what actually gets persisted; validate that shape.
51
+ payload = json.loads(json.dumps(payload, default=str))
52
+ err = _validate(payload, 'schemas/vNext/negative-search-record.schema.json',
53
+ 'negative-search-record')
54
+ if err:
55
+ problems.append(err)
56
+ checked += 1
57
+ except Exception as exc: # import or construction failure is itself a defect
58
+ problems.append(f'negative-search-record: could not build record: {exc}')
59
+
60
+ # --- vNext: autoresearch lifecycle records ---------------------------
61
+ # Every schema in this family is exercised against the dataclass that
62
+ # actually produces it, so a renamed field, retyped value or narrowed
63
+ # enum fails CI instead of drifting silently. Only `negative-search-record`
64
+ # was wired before; the rest parsed as JSON and nothing more.
65
+ try:
66
+ import dataclasses
67
+
68
+ from engine.autoresearch.contracts import (ResearchBudget,
69
+ ResearchExperimentType,
70
+ ResearchIteration,
71
+ ResearchStrategy,
72
+ IterationStatus)
73
+ from engine.autoresearch.gap_priority import GapPriority
74
+
75
+ strategy = ResearchStrategy(
76
+ strategy_id='RST-1',
77
+ experiment_type=ResearchExperimentType.TARGETED_RETRIEVAL,
78
+ hypothesis='a targeted retrieval closes the retention gap',
79
+ expected_gain='one or more direct retention findings',
80
+ budget=ResearchBudget())
81
+ iteration = ResearchIteration(
82
+ iteration_id='RI-1', project_id='PRJ-1', base_graph_revision=1,
83
+ gap_id='GAP-1', strategy=strategy)
84
+ iteration.complete(IterationStatus.SEARCH_SATURATED)
85
+ iteration_payload = json.loads(json.dumps(iteration.as_dict(), default=str))
86
+ strategy_payload = json.loads(json.dumps(dataclasses.asdict(strategy), default=str))
87
+ strategy_payload['experiment_type'] = strategy.experiment_type.value
88
+ priority = GapPriority(
89
+ gap_id='GAP-1', dvi_band='HIGH', cost_band='LOW',
90
+ decision_material=True, drivers=('missing_retention',),
91
+ # The enum is the controller's own vocabulary (gap_priority.py and
92
+ # controller.py emit these three); it is not the experiment-type
93
+ # vocabulary. Using a plausible-looking but wrong value here is
94
+ # exactly the drift this check exists to catch.
95
+ next_research_mode='secondary_evidence_search', score=1)
96
+ priority_payload = json.loads(json.dumps(dataclasses.asdict(priority), default=str))
97
+
98
+ for label, payload, schema_rel in (
99
+ ('research-strategy', strategy_payload, 'schemas/vNext/research-strategy.schema.json'),
100
+ ('research-iteration', iteration_payload, 'schemas/vNext/research-iteration.schema.json'),
101
+ ('gap-priority', priority_payload, 'schemas/vNext/gap-priority.schema.json'),
102
+ ):
103
+ err = _validate(payload, schema_rel, label)
104
+ if err:
105
+ problems.append(err)
106
+ checked += 1
107
+ except Exception as exc:
108
+ problems.append(f'autoresearch lifecycle records: could not build record: {exc}')
109
+
110
+ # --- vNext: orchestration plan and task contracts ---------------------
111
+ try:
112
+ import dataclasses as _dc
113
+
114
+ from engine.orchestration import ExecutionPlanner, ExecutionMode
115
+
116
+ # A delegated plan needs a run id and a base revision; without them
117
+ # TaskSpec.validate_for_dispatch refuses to describe a dispatchable task.
118
+ plan = ExecutionPlanner().plan('M', run_id='RUN-1', base_revision=1)
119
+ plan_payload = json.loads(json.dumps(_dc.asdict(plan), default=str))
120
+ plan_payload['complexity'] = str(getattr(plan.complexity, 'value', plan.complexity))
121
+ err = _validate(plan_payload, 'schemas/vNext/execution-plan.schema.json',
122
+ 'execution-plan')
123
+ if err:
124
+ problems.append(err)
125
+ checked += 1
126
+
127
+ delegated = [t for t in plan.tasks if t.execution_mode is ExecutionMode.DELEGATED]
128
+ if not delegated:
129
+ problems.append('execution plan has no delegated task to exercise task-spec')
130
+ else:
131
+ task_payload = json.loads(json.dumps(delegated[0].to_dict(), default=str))
132
+ err = _validate(task_payload, 'schemas/vNext/task-spec.schema.json', 'task-spec')
133
+ if err:
134
+ problems.append(err)
135
+ checked += 1
136
+
137
+ # A WorkerResult is built only by the lead process, which ignores
138
+ # worker self-attestation; exercise that real path, not a hand-built dict.
139
+ from engine.worker_result import validate_worker_output
140
+ result = validate_worker_output(delegated[0], {
141
+ 'task_id': delegated[0].task_id,
142
+ 'status': 'completed',
143
+ 'staging_artifacts': [{'artifact_type': delegated[0].expected_outputs[0],
144
+ 'summary': 'staged'}],
145
+ 'validated': True,
146
+ 'metrics': {},
147
+ 'summary': 'ok',
148
+ })
149
+ err = _validate(json.loads(json.dumps(result.to_dict(), default=str)),
150
+ 'schemas/vNext/worker-result.schema.json', 'worker-result')
151
+ if err:
152
+ problems.append(err)
153
+ checked += 1
154
+ except Exception as exc:
155
+ problems.append(f'orchestration records: could not build record: {exc}')
156
+
157
+ # --- vNext: autoevolve session and experiment records -----------------
158
+ try:
159
+ import dataclasses as _dc2
160
+
161
+ from engine.autoevolve.core import EvalSnapshot, SkillExperiment
162
+
163
+ snapshot = EvalSnapshot(
164
+ eval_id='EVAL-1', hard_gates_passed=True, science_score=0.8,
165
+ research_score=0.7, robustness=0.9, cost=0.0,
166
+ latency=12.5, complexity=0.5, repeats=3, noise_floor=0.01,
167
+ dev_passed=True, holdout_passed=True, adversarial_passed=True,
168
+ holdout_isolation_verified=False, eval_suite_hash='deadbeef')
169
+ snapshot_payload = json.loads(json.dumps(_dc2.asdict(snapshot), default=str))
170
+ err = _validate(snapshot_payload, 'schemas/vNext/eval-snapshot.schema.json',
171
+ 'eval-snapshot')
172
+ if err:
173
+ problems.append(err)
174
+ checked += 1
175
+
176
+ experiment = SkillExperiment(
177
+ experiment_id='EXP-1', session_id='session-1',
178
+ parent_skill_revision='rev-1', hypothesis='a narrower prompt scores higher',
179
+ mutation_scope=('safe',), changed_files=('skill/agents/skeptic.md',),
180
+ candidate_commit='deadbeef', baseline_eval_id='EVAL-1',
181
+ candidate_eval_id='EVAL-2', protected_hash_before='h1',
182
+ protected_hash_after='h1', status='REJECT',
183
+ promotion_reason='no significant improvement')
184
+ experiment_payload = json.loads(json.dumps(_dc2.asdict(experiment), default=str))
185
+ err = _validate(experiment_payload, 'schemas/vNext/skill-experiment.schema.json',
186
+ 'skill-experiment')
187
+ if err:
188
+ problems.append(err)
189
+ checked += 1
190
+
191
+ # The session report is the real runner payload (runner.py writes
192
+ # daily-report.json); validate that documented shape directly.
193
+ session_report = {
194
+ 'run_tag': 'session-1', 'branch': 'autoevolve/session-1',
195
+ 'experiments': 1, 'statuses': ['REJECT'], 'best_experiment_id': None,
196
+ 'best_candidate_commit': None, 'cost': 0.0, 'wall_minutes': 1.0,
197
+ 'plateau': False, 'stop_reason': 'completed',
198
+ 'promotion': 'branch_only', 'branch_push_requested': False,
199
+ 'branch_pushed': False, 'mutation_view': 'dev_only_context_isolation',
200
+ 'holdout_isolation_verified': False, 'isolation_provider': 'none',
201
+ 'isolation_reason': 'no os isolation provider available',
202
+ 'eval_suite_hash': 'deadbeef', 'security_note': 'branch only',
203
+ 'candidate_artifacts': 'local session state only; never auto-pushed',
204
+ }
205
+ err = _validate(session_report, 'schemas/vNext/autoevolve-session.schema.json',
206
+ 'autoevolve-session')
207
+ if err:
208
+ problems.append(err)
209
+ checked += 1
210
+ except Exception as exc:
211
+ problems.append(f'autoevolve records: could not build record: {exc}')
212
+
213
+ # --- v3: the BENCHMARK run manifest (scripts/benchmark_v3.py) ---------
214
+ # Note: schemas/v3/run-manifest.schema.json describes the empirical
215
+ # benchmark harness record, not the run-workspace manifest. Validating the
216
+ # wrong producer is how a contract silently stops describing reality.
217
+ try:
218
+ fixture = sorted((ROOT / 'benchmarks' / 'empirical').glob('*/manifest.json'))
219
+ if fixture:
220
+ payload = json.loads(fixture[0].read_text(encoding='utf-8'))
221
+ err = _validate(payload, 'schemas/v3/run-manifest.schema.json',
222
+ f'run-manifest ({fixture[0].parent.name})')
223
+ if err:
224
+ problems.append(err)
225
+ checked += 1
226
+ else:
227
+ # benchmarks/empirical/ holds local empirical runs and is excluded
228
+ # from the submission package on purpose. Its absence means there is
229
+ # nothing to check in this checkout, not that the contract is broken.
230
+ print('note: no benchmarks/empirical/*/manifest.json in this checkout; '
231
+ 'skipping the v3 run-manifest data check')
232
+ except Exception as exc:
233
+ problems.append(f'v3 run-manifest: {exc}')
234
+
235
+ # --- sanity: every declared schema still parses -----------------------
236
+ for family in ('v3', 'v4', 'vNext'):
237
+ for path in sorted((ROOT / 'schemas' / family).glob('*.json')):
238
+ try:
239
+ json.loads(path.read_text(encoding='utf-8'))
240
+ checked += 1
241
+ except (OSError, json.JSONDecodeError) as exc:
242
+ problems.append(f'{path.relative_to(ROOT)}: {exc}')
243
+
244
+ if problems:
245
+ print('ERROR: versioned schema validation failed', file=sys.stderr)
246
+ for item in problems:
247
+ print(f' {item}', file=sys.stderr)
248
+ return 1
249
+ print(f'versioned schemas OK ({checked} checks across v3 / v4 / vNext)')
250
+ return 0
251
+
252
+
253
+ if __name__ == '__main__':
254
+ sys.exit(main())
@@ -27,14 +27,19 @@ from pathlib import Path
27
27
 
28
28
  from evidence_semantics import claim_relation
29
29
 
30
- SUPPORTED_OUTCOMES = {
31
- "knowledge_gain", "concept_understanding", "retention", "transfer",
32
- "independent_problem_solving", "completion_time", "accuracy",
33
- "code_quality", "assignment_score", "engagement", "motivation",
34
- "cognitive_load", "help_seeking", "metacognition", "ai_dependency",
35
- "over_reliance", "reduced_effort", "reduced_transfer",
36
- "academic_integrity_risk", "false_confidence",
37
- }
30
+ def _supported_outcomes() -> set[str]:
31
+ """Every registered outcome token, read from the domain registry.
32
+
33
+ This was a hand-copied 20-token education list in three separate files;
34
+ it silently rejected policy tokens such as policy_effectiveness. The
35
+ registry (domains/<id>/outcome_taxonomy.json) is the single authority.
36
+ """
37
+ from engine.taxonomy import all_tokens_ordered
38
+
39
+ return set(all_tokens_ordered())
40
+
41
+
42
+ SUPPORTED_OUTCOMES = _supported_outcomes()
38
43
 
39
44
 
40
45
  def load_records(path: Path) -> list[dict]:
@@ -40,6 +40,16 @@ import json
40
40
  import sys
41
41
  from pathlib import Path
42
42
 
43
+ # This module is also documented as a standalone command (SKILL.md), so it
44
+ # has to find the repository root on its own path; the orchestrator sets
45
+ # sys.path for it, which hid the missing import from that caller.
46
+ import sys as _sys
47
+ from pathlib import Path as _Path
48
+ _ROOT = _Path(__file__).resolve().parent.parent
49
+ if str(_ROOT) not in _sys.path:
50
+ _sys.path.insert(0, str(_ROOT))
51
+
52
+
43
53
  from evidence_score import (CONFIDENCE_POLICY_VERSION,
44
54
  decision_consistency_score, directness_score,
45
55
  independent_samples, independent_studies)
@@ -524,7 +524,8 @@ class StudioHandler(http.server.SimpleHTTPRequestHandler):
524
524
  self._send_report_bytes(html_path.read_bytes())
525
525
 
526
526
 
527
- def run_dashboard_server(host: str = "127.0.0.1", port: int = 8765) -> None:
527
+ def run_dashboard_server(host: str = "127.0.0.1", port: int = 8765,
528
+ open_browser: bool = False) -> None:
528
529
  server = None
529
530
  actual_port = port
530
531
  for offset in range(10):
@@ -541,12 +542,20 @@ def run_dashboard_server(host: str = "127.0.0.1", port: int = 8765) -> None:
541
542
  if server is None:
542
543
  print(f"❌ 端口 {port}-{port + 9} 均被占用。")
543
544
  return
545
+ studio_url = f"http://{host}:{actual_port}/studio/"
544
546
  print("============================================================")
545
547
  print(f"🚀 EduEvidence Web Studio running at http://{host}:{actual_port}/")
546
548
  print(" Research Studio /studio/ (read-only)")
547
549
  print(" Projects, evidence, revisions and five-theme reports")
548
550
  print(" Projection API /api/studio/catalog")
549
551
  print("============================================================")
552
+ if open_browser:
553
+ try:
554
+ import webbrowser
555
+ webbrowser.open(studio_url)
556
+ print(f"🌐 opened {studio_url}")
557
+ except Exception as exc: # pragma: no cover - browser may be absent
558
+ print(f"⚠️ failed to open browser: {exc}")
550
559
  try:
551
560
  server.serve_forever()
552
561
  except KeyboardInterrupt:
@@ -557,8 +566,10 @@ def main() -> None:
557
566
  parser = argparse.ArgumentParser()
558
567
  parser.add_argument("--host", default="127.0.0.1")
559
568
  parser.add_argument("--port", type=int, default=8765)
569
+ parser.add_argument("--open", action="store_true",
570
+ help="open the Studio URL in the system browser once the server is up")
560
571
  args = parser.parse_args()
561
- run_dashboard_server(args.host, args.port)
572
+ run_dashboard_server(args.host, args.port, open_browser=args.open)
562
573
 
563
574
 
564
575
  if __name__ == "__main__":
@@ -102,12 +102,22 @@ def run_did_analysis(csv_path: str) -> Dict[str, Any]:
102
102
  cl = col.strip().lower()
103
103
  if cl in ("cluster_id", "class_id", "school_id", "group_id") or cl.endswith("_cluster"):
104
104
  cluster_columns.append(col)
105
+ # Order matters. An outcome column is often named "post_test_score",
106
+ # which also contains "post": matching the period rule first stole the
107
+ # outcome column and the run failed with ERR_MISSING_COLUMNS. Outcome
108
+ # and treatment are the more specific patterns, so they are tested
109
+ # before the period keyword.
105
110
  if "treat" in cl or cl in ("group", "condition", "is_treatment"):
106
111
  field_map["treat"] = col
107
- elif "post" in cl or "after" in cl or "period" in cl or "time" in cl or "pre_post" in cl:
112
+ elif cl in ("post", "posttest", "pre_post", "period", "time_period"):
108
113
  field_map["post"] = col
109
- elif "score" in cl or "outcome" in cl or "grade" in cl or "result" in cl or "performance" in cl or cl == "y":
114
+ elif ("score" in cl or "outcome" in cl or "grade" in cl or "result" in cl
115
+ or "performance" in cl or cl == "y"):
110
116
  field_map["outcome"] = col
117
+ elif "post" in cl or "after" in cl or "period" in cl or "time" in cl:
118
+ # Generic period/phase column, only after the specific patterns
119
+ # above have had their chance.
120
+ field_map.setdefault("post", col)
111
121
 
112
122
  if "treat" not in field_map or "post" not in field_map or "outcome" not in field_map:
113
123
  return {
@@ -36,8 +36,11 @@ from evidence_semantics import claim_relation, decision_relation
36
36
  DIMENSIONS = ["D1_study_design", "D2_sample_quality", "D3_measurement_validity",
37
37
  "D4_temporal_strength", "D5_directness"]
38
38
 
39
- #: Version of the deterministic confidence policy (bump on any formula change).
40
- CONFIDENCE_POLICY_VERSION = "2026-08-12.v2"
39
+ #: Version of the deterministic confidence policy. Imported from the engine so
40
+ #: there is one authority: evidence_score previously hard-coded ".v2" while
41
+ #: engine/versions.py declared ".v3", so verdicts recorded a policy version the
42
+ #: engine did not recognise and nothing compared the two.
43
+ from engine.versions import CONFIDENCE_POLICY_VERSION # noqa: F401 (re-export)
41
44
 
42
45
 
43
46
  def quality_score(dimensions: dict[str, int]) -> float:
@@ -0,0 +1,31 @@
1
+ """intake — one-shot two-round user intake + durable prefs + report open."""
2
+ from __future__ import annotations
3
+
4
+ from .browser import (
5
+ maybe_open_after_run,
6
+ open_report,
7
+ open_url,
8
+ resolve_main_report,
9
+ )
10
+ from .constants import (
11
+ DEPTH_CHOICES,
12
+ ENHANCEMENTS,
13
+ ENHANCEMENT_BLURBS,
14
+ DEPTH_BLURBS,
15
+ FRAME_SLOTS,
16
+ THEME_NAMES,
17
+ )
18
+ from .depth import auto_resolve_depth, resolve_depth
19
+ from .background import infer_frame_hints
20
+ from .prefs import default_prefs, load_prefs, prefs_path, save_prefs
21
+ from .prompts import PROMPT_SCRIPT
22
+ from .session import run_intake
23
+
24
+ __all__ = [
25
+ "DEPTH_BLURBS", "DEPTH_CHOICES", "ENHANCEMENTS", "ENHANCEMENT_BLURBS",
26
+ "FRAME_SLOTS", "PROMPT_SCRIPT", "THEME_NAMES",
27
+ "auto_resolve_depth", "default_prefs", "infer_frame_hints",
28
+ "load_prefs", "maybe_open_after_run", "open_report", "open_url",
29
+ "prefs_path", "resolve_depth", "resolve_main_report", "run_intake",
30
+ "save_prefs",
31
+ ]
@@ -0,0 +1,18 @@
1
+ """python -m intake — CLI entry."""
2
+ from __future__ import annotations
3
+
4
+ import sys
5
+ from pathlib import Path
6
+
7
+ # Allow `python scripts/intake/__main__.py` without installing the package.
8
+ if __package__ in (None, ""):
9
+ sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
10
+
11
+ from intake.cli import main # noqa: E402
12
+
13
+ if __name__ == "__main__":
14
+ try:
15
+ raise SystemExit(main())
16
+ except ValueError as exc:
17
+ print(f"ERROR: {exc}", file=sys.stderr)
18
+ raise SystemExit(2) from exc
@@ -0,0 +1,78 @@
1
+ """Frame skeleton inference + 3–6 boundary questions (never fabricate)."""
2
+ from __future__ import annotations
3
+
4
+ import re
5
+ from typing import Any, Callable
6
+
7
+ from .constants import (
8
+ CONTEXT_CUES,
9
+ FRAME_SLOTS,
10
+ OUTCOME_CUES,
11
+ POP_CUES,
12
+ SLOT_LABELS,
13
+ )
14
+ from .prompts import ask, ask_yes
15
+
16
+
17
+ def infer_frame_hints(question: str, domain: str = "education") -> dict[str, Any]:
18
+ """Propose Frame boundary hints from the question text only.
19
+
20
+ Anything not stated in the question stays absent. Never fabricate.
21
+ """
22
+ hints: dict[str, Any] = {"domain": domain, "confirmed": False}
23
+ text = question or ""
24
+ low = text.lower()
25
+ for cue in POP_CUES:
26
+ if cue.lower() in low:
27
+ hints["population"] = cue
28
+ break
29
+ for cue in OUTCOME_CUES:
30
+ if cue.lower() in low:
31
+ hints["primary_outcome"] = cue
32
+ break
33
+ for cue in CONTEXT_CUES:
34
+ if cue in text or cue.lower() in low:
35
+ hints["context"] = cue
36
+ break
37
+ m = re.search(r"(AI[\w\s一-鿿]{0,20}(?:工具|助手|tutor|copilot|编程)|"
38
+ r"(?:翻转|项目式|同伴)教学|AI\s*编程助手)", text, re.I)
39
+ if m:
40
+ hints["intervention"] = m.group(0).strip()
41
+ if re.search(r"是否|影响|对比|比较|vs|与.*比|是否允许|该不该", text):
42
+ hints["comparison"] = "business as usual / 未使用该干预的常规做法"
43
+ return hints
44
+
45
+
46
+ def boundary_questions(inferred: dict[str, Any]) -> list[tuple[str, str, str | None]]:
47
+ """3–6 boundary questions: (slot, prompt, default)."""
48
+ questions: list[tuple[str, str, str | None]] = []
49
+ for slot in FRAME_SLOTS:
50
+ questions.append((slot, f" · {SLOT_LABELS[slot]}", inferred.get(slot)))
51
+ return questions[:6]
52
+
53
+
54
+ def collect_frame_hints(question: str, *, domain: str = "education",
55
+ interactive: bool,
56
+ printer: Callable[[str], None] = print) -> dict[str, Any]:
57
+ """Infer skeleton; when interactive, ask 3–6 boundary Qs + one confirm."""
58
+ inferred = infer_frame_hints(question, domain=domain)
59
+ frame_hints = dict(inferred)
60
+ if not interactive:
61
+ frame_hints["confirmed"] = False
62
+ return frame_hints
63
+
64
+ printer("—— 背景智能深挖(Frame 边界,智能默认 + 一次确认;缺失不编造)——")
65
+ printer(f" 领域骨架:{domain}")
66
+ for slot, prompt, default in boundary_questions(inferred):
67
+ label = default if default is not None else "(未从题目推出,留空)"
68
+ raw = ask(prompt, label if default is not None else "")
69
+ if raw and not raw.startswith("("):
70
+ frame_hints[slot] = raw
71
+ elif default is not None:
72
+ frame_hints[slot] = default
73
+ # else: leave absent — never fabricate
74
+ printer(" Frame 边界摘要:")
75
+ for slot in FRAME_SLOTS:
76
+ printer(f" {slot}: {frame_hints.get(slot) or '—'}")
77
+ frame_hints["confirmed"] = ask_yes(" 确认以上边界?(Y/n)", True)
78
+ return frame_hints
@@ -0,0 +1,79 @@
1
+ """Main-report resolution and optional system-browser open."""
2
+ from __future__ import annotations
3
+
4
+ import webbrowser
5
+ from pathlib import Path
6
+
7
+ from .constants import THEME_NAMES
8
+ from .prefs import load_prefs
9
+ from .prompts import is_interactive
10
+
11
+
12
+ def resolve_main_report(directory: Path | str, theme: str | None = None) -> Path | None:
13
+ """Pick the main report file. All five themes stay on disk.
14
+
15
+ Preference order:
16
+ reports-5themes/EduEvidence_Report_<theme>.html
17
+ reports-5themes/EduEvidence_Report_claude.html
18
+ report.html
19
+ EduEvidence_Report.html
20
+ """
21
+ root = Path(directory)
22
+ if not root.is_dir():
23
+ return None
24
+ theme = theme if theme in THEME_NAMES else "claude"
25
+ variants = root / "reports-5themes"
26
+ candidates = [
27
+ variants / f"EduEvidence_Report_{theme}.html",
28
+ variants / "EduEvidence_Report_claude.html",
29
+ root / "report.html",
30
+ root / "EduEvidence_Report.html",
31
+ ]
32
+ for path in candidates:
33
+ if path.is_file() and path.stat().st_size > 2:
34
+ return path
35
+ return None
36
+
37
+
38
+ def open_report(directory: Path | str, theme: str | None = None,
39
+ *, force: bool = False) -> bool:
40
+ """Open the preferred main report in the system browser.
41
+
42
+ Only opens when `force` or prefs.open_browser is set. The caller should
43
+ also suppress this on non-TTY if silent behaviour is required.
44
+ """
45
+ prefs = load_prefs()
46
+ if not force and not prefs.get("open_browser", True):
47
+ return False
48
+ path = resolve_main_report(directory, theme or prefs.get("default_main_theme"))
49
+ if path is None:
50
+ return False
51
+ try:
52
+ webbrowser.open(path.resolve().as_uri())
53
+ except Exception:
54
+ return False
55
+ return True
56
+
57
+
58
+ def open_url(url: str, *, force: bool = False) -> bool:
59
+ prefs = load_prefs()
60
+ if not force and not prefs.get("open_browser", True):
61
+ return False
62
+ try:
63
+ webbrowser.open(url)
64
+ except Exception:
65
+ return False
66
+ return True
67
+
68
+
69
+ def maybe_open_after_run(run_dir: Path | str, *, theme: str | None = None,
70
+ interactive: bool | None = None) -> bool:
71
+ """End-of-run auto-open. Silent on non-TTY so tests stay quiet."""
72
+ if interactive is None:
73
+ interactive = is_interactive()
74
+ if not interactive:
75
+ return False
76
+ prefs = load_prefs()
77
+ if not prefs.get("open_browser", True):
78
+ return False
79
+ return open_report(run_dir, theme or prefs.get("default_main_theme"), force=True)
@@ -0,0 +1,57 @@
1
+ """CLI for the one-shot two-round intake."""
2
+ from __future__ import annotations
3
+
4
+ import argparse
5
+ import json
6
+ import sys
7
+
8
+ from .constants import DEPTH_CHOICES, ENHANCEMENTS
9
+ from .prefs import load_prefs
10
+ from .prompts import PROMPT_SCRIPT
11
+ from .session import run_intake
12
+
13
+
14
+ def main(argv: list[str] | None = None) -> int:
15
+ parser = argparse.ArgumentParser(description="EduEvidence two-round intake")
16
+ parser.add_argument("--question", default=None, help="research question")
17
+ parser.add_argument("--depth", default=None,
18
+ choices=list(DEPTH_CHOICES) + ["quick", "standard", "deep"],
19
+ help="S/M/L/auto (or legacy quick/standard/deep)")
20
+ parser.add_argument("--enhancement", action="append", default=None,
21
+ choices=list(ENHANCEMENTS),
22
+ help="execution enhancement (repeatable)")
23
+ parser.add_argument("--domain", default="education")
24
+ parser.add_argument("--yes", action="store_true",
25
+ help="skip prompts; use prefs + explicit flags (unattended)")
26
+ parser.add_argument("--print-prompts", action="store_true",
27
+ help="print the fixed two-round prompt script and exit")
28
+ parser.add_argument("--show-prefs", action="store_true",
29
+ help="print current prefs.json and exit")
30
+ args = parser.parse_args(argv)
31
+
32
+ if args.print_prompts:
33
+ sys.stdout.write(PROMPT_SCRIPT)
34
+ return 0
35
+ if args.show_prefs:
36
+ print(json.dumps(load_prefs(), ensure_ascii=False, indent=2))
37
+ return 0
38
+
39
+ record = run_intake(
40
+ question=args.question,
41
+ depth=args.depth,
42
+ enhancements=args.enhancement,
43
+ domain=args.domain,
44
+ assume_yes=args.yes,
45
+ )
46
+ print(json.dumps(record, ensure_ascii=False, indent=2))
47
+ return 0
48
+
49
+
50
+ if __name__ == "__main__":
51
+ # Documented entry: `python3 -m intake.cli --print-prompts` from scripts/.
52
+ # Same failure contract as scripts/intake/__main__.py.
53
+ try:
54
+ raise SystemExit(main())
55
+ except ValueError as exc:
56
+ print(f"ERROR: {exc}", file=sys.stderr)
57
+ raise SystemExit(2) from exc