eduevidence 5.2.0 → 6.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (276) hide show
  1. package/README.md +54 -42
  2. package/README.zh-CN.md +51 -28
  3. package/SKILL.md +390 -133
  4. package/agents/openai.yaml +4 -0
  5. package/assets/readme/controlled-execution.svg +34 -0
  6. package/assets/readme/logo.png +0 -0
  7. package/assets/readme/research-workflow.svg +56 -0
  8. package/assets/readme/studio-graph.png +0 -0
  9. package/assets/readme/studio-overview.png +0 -0
  10. package/assets/readme/studio-reports.png +0 -0
  11. package/autoevolve/config.yaml +17 -0
  12. package/autoevolve/program.md +25 -0
  13. package/autoevolve/protected.manifest.yaml +34 -0
  14. package/benchmarks/adversarial/cases.jsonl +7 -0
  15. package/benchmarks/evidence-library.json +5268 -0
  16. package/benchmarks/partitions.json +8 -0
  17. package/docs/architecture.md +220 -0
  18. package/docs/autoresearch-evolution-plan.md +2903 -0
  19. package/docs/autoresearch-implementation-status.md +101 -0
  20. package/docs/demo-storyboard.md +20 -0
  21. package/docs/demo-workplace-ai.md +92 -0
  22. package/docs/demo.md +32 -0
  23. package/docs/install-guide.md +150 -0
  24. package/docs/orchestration-role-model.md +1254 -0
  25. package/docs/release-closeout/README.md +17 -0
  26. package/docs/release-closeout/frontend-acceptance.md +23 -0
  27. package/docs/release-closeout/issues.md +19 -0
  28. package/docs/release-closeout/verification.md +28 -0
  29. package/docs/release-contract.md +108 -0
  30. package/docs/research-studio-guide.zh-CN.md +166 -0
  31. package/eduevidence_cli.py +17 -11
  32. package/engine/_resources.py +13 -0
  33. package/engine/autoevolve/__init__.py +3 -0
  34. package/engine/autoevolve/agent_view.py +167 -0
  35. package/engine/autoevolve/core.py +357 -0
  36. package/engine/autoevolve/events.py +11 -0
  37. package/engine/autoevolve/git_workspace.py +77 -0
  38. package/engine/autoevolve/projection.py +23 -0
  39. package/engine/autoevolve/runner.py +413 -0
  40. package/engine/autoevolve/trust.py +146 -0
  41. package/engine/autoresearch/__init__.py +6 -0
  42. package/engine/autoresearch/commit.py +132 -0
  43. package/engine/autoresearch/contracts.py +126 -0
  44. package/engine/autoresearch/controller.py +207 -0
  45. package/engine/autoresearch/events.py +12 -0
  46. package/engine/autoresearch/gap_priority.py +168 -0
  47. package/engine/autoresearch/projection.py +30 -0
  48. package/engine/autoresearch/research_memory.py +59 -0
  49. package/engine/autoresearch/saturation.py +91 -0
  50. package/engine/briefs.py +2 -1
  51. package/engine/capabilities.py +1 -0
  52. package/engine/contracts.py +3 -1
  53. package/engine/evidencecore.py +7 -5
  54. package/engine/gaps.py +90 -51
  55. package/engine/judge_pack.py +65 -0
  56. package/engine/library_builtin.py +3 -1
  57. package/engine/living.py +2 -1
  58. package/engine/meta_synthesis.py +3 -1
  59. package/engine/orchestration.py +460 -0
  60. package/engine/pilot.py +2 -1
  61. package/engine/project.py +2 -2
  62. package/engine/research_service.py +113 -0
  63. package/engine/studio_read_model.py +400 -0
  64. package/engine/tribunal.py +1 -2
  65. package/engine/update.py +1 -0
  66. package/engine/versions.py +1 -1
  67. package/engine/worker_result.py +109 -0
  68. package/engine/workflows.py +70 -0
  69. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1720 -0
  70. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
  71. package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
  72. package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
  73. package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
  74. package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
  75. package/examples/ai-coding-assistant-evidence/frame.json +48 -0
  76. package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
  77. package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
  78. package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
  79. package/examples/ai-coding-assistant-evidence/report_spec.json +219 -0
  80. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2614 -0
  81. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2614 -0
  82. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2614 -0
  83. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2614 -0
  84. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2614 -0
  85. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1720 -0
  86. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1720 -0
  87. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1720 -0
  88. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1720 -0
  89. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1720 -0
  90. package/examples/ai-coding-assistant-evidence/result.json +1453 -0
  91. package/examples/ai-coding-assistant-evidence/result.zh.json +1453 -0
  92. package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
  93. package/examples/ai-coding-assistant-evidence/verdict.json +103 -0
  94. package/examples/workplace-ai-assistant/claims.jsonl +4 -0
  95. package/examples/workplace-ai-assistant/evaluation.json +19 -0
  96. package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
  97. package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
  98. package/examples/workplace-ai-assistant/frame.json +41 -0
  99. package/examples/workplace-ai-assistant/intervention.json +27 -0
  100. package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
  101. package/examples/workplace-ai-assistant/methodology.json +60 -0
  102. package/examples/workplace-ai-assistant/report_spec.json +55 -0
  103. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2484 -0
  104. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2484 -0
  105. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2484 -0
  106. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2484 -0
  107. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2484 -0
  108. package/examples/workplace-ai-assistant/result.json +553 -0
  109. package/examples/workplace-ai-assistant/result.zh.json +553 -0
  110. package/examples/workplace-ai-assistant/search_log.json +19 -0
  111. package/examples/workplace-ai-assistant/sources.jsonl +3 -0
  112. package/examples/workplace-ai-assistant/validation_result.json +9 -0
  113. package/examples/workplace-ai-assistant/verdict.json +52 -0
  114. package/install.sh +7 -7
  115. package/integrations/orchestration_dispatch.py +146 -0
  116. package/package.json +37 -3
  117. package/pyproject.toml +11 -20
  118. package/references/autoresearch.md +30 -0
  119. package/references/evaluation-policy.md +24 -0
  120. package/references/orchestration.md +22 -0
  121. package/references/scientific-invariants.md +19 -0
  122. package/retrieval/audit.py +154 -0
  123. package/schemas/intervention.schema.json +106 -21
  124. package/schemas/report-result.schema.json +9 -1
  125. package/schemas/v2/project.schema.json +2 -2
  126. package/schemas/v2/run.schema.json +1 -1
  127. package/schemas/vNext/autoevolve-session.schema.json +1 -0
  128. package/schemas/vNext/eval-snapshot.schema.json +1 -0
  129. package/schemas/vNext/execution-plan.schema.json +1 -0
  130. package/schemas/vNext/gap-priority.schema.json +1 -0
  131. package/schemas/vNext/negative-search-record.schema.json +1 -0
  132. package/schemas/vNext/research-iteration.schema.json +1 -0
  133. package/schemas/vNext/research-strategy.schema.json +1 -0
  134. package/schemas/vNext/skill-experiment.schema.json +1 -0
  135. package/schemas/vNext/task-spec.schema.json +1 -0
  136. package/schemas/vNext/worker-result.schema.json +1 -0
  137. package/scripts/benchmark_judge.py +2 -2
  138. package/scripts/benchmark_v3.py +26 -43
  139. package/scripts/build_esl_artifacts.py +2 -2
  140. package/scripts/build_evidence_library.py +2 -2
  141. package/scripts/build_gh_pages.py +98 -0
  142. package/scripts/build_readme_diagrams.py +72 -0
  143. package/scripts/build_report_variants.py +85 -0
  144. package/scripts/check_autoresearch_invariants.py +95 -0
  145. package/scripts/daily_evolve.py +30 -0
  146. package/scripts/dashboard_server.py +130 -101
  147. package/scripts/did_regression.py +5 -30
  148. package/scripts/enrich_projects_human_and_lieflat.py +1 -1
  149. package/scripts/generate_metrics.py +4 -3
  150. package/scripts/generate_new_projects.py +1 -1
  151. package/scripts/orchestrator.py +172 -18
  152. package/scripts/rebake_all_5themes.py +1 -2
  153. package/scripts/research_auto_cli.py +475 -0
  154. package/scripts/run_workspace.py +17 -7
  155. package/scripts/search_provenance.py +64 -0
  156. package/scripts/serve_web.py +9 -10
  157. package/scripts/skill_lint.py +1 -1
  158. package/scripts/skill_payload.py +78 -0
  159. package/scripts/validate_schema.py +15 -1
  160. package/scripts/vnext_cli.py +133 -0
  161. package/setup.py +12 -0
  162. package/skill/roles/registry.yaml +45 -0
  163. package/skill/sub-skills/report-generation/SKILL.md +12 -6
  164. package/skill/task-briefs/applicability.md +3 -0
  165. package/skill/task-briefs/projection.md +3 -0
  166. package/skill/workflows/decision-and-pilot.md +10 -0
  167. package/skill/workflows/evaluate-and-update.md +10 -0
  168. package/skill/workflows/evidence-review.md +13 -0
  169. package/visualization/eduevidence-report/assets/base.css +2 -2
  170. package/visualization/eduevidence-report/assets/reader.css +752 -0
  171. package/visualization/eduevidence-report/assets/reader.js +132 -0
  172. package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
  173. package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
  174. package/visualization/eduevidence-report/scripts/build_report.py +58 -65
  175. package/visualization/eduevidence-report/scripts/lieflat_engine.py +24 -100
  176. package/visualization/eduevidence-report/themes/academic.css +1 -1
  177. package/visualization/eduevidence-report/themes/claude.css +1 -1
  178. package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
  179. package/visualization/eduevidence-report/themes/datalab.css +2 -2
  180. package/visualization/eduevidence-report/themes/presentation.css +2 -2
  181. package/web/README.md +18 -0
  182. package/web/index.html +53 -0
  183. package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
  184. package/web/studio/assets/index-CzXocaGv.css +1 -0
  185. package/web/studio/assets/index-pa7jD7n4.js +230 -0
  186. package/web/studio/config.json +1 -0
  187. package/web/studio/index.html +14 -0
  188. package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
  189. package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
  190. package/engine/__pycache__/bias.cpython-312.pyc +0 -0
  191. package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
  192. package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
  193. package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
  194. package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
  195. package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
  196. package/engine/__pycache__/events.cpython-312.pyc +0 -0
  197. package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
  198. package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
  199. package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
  200. package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
  201. package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
  202. package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
  203. package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
  204. package/engine/__pycache__/ids.cpython-312.pyc +0 -0
  205. package/engine/__pycache__/library.cpython-312.pyc +0 -0
  206. package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
  207. package/engine/__pycache__/living.cpython-312.pyc +0 -0
  208. package/engine/__pycache__/log.cpython-312.pyc +0 -0
  209. package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
  210. package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
  211. package/engine/__pycache__/migration.cpython-312.pyc +0 -0
  212. package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
  213. package/engine/__pycache__/paths.cpython-312.pyc +0 -0
  214. package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
  215. package/engine/__pycache__/planner.cpython-312.pyc +0 -0
  216. package/engine/__pycache__/project.cpython-312.pyc +0 -0
  217. package/engine/__pycache__/projections.cpython-312.pyc +0 -0
  218. package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
  219. package/engine/__pycache__/run.cpython-312.pyc +0 -0
  220. package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
  221. package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
  222. package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
  223. package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
  224. package/engine/__pycache__/update.cpython-312.pyc +0 -0
  225. package/engine/__pycache__/versions.cpython-312.pyc +0 -0
  226. package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
  227. package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
  228. package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
  229. package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
  230. package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
  231. package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
  232. package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
  233. package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
  234. package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
  235. package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
  236. package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
  237. package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
  238. package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
  239. package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
  240. package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
  241. package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
  242. package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
  243. package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
  244. package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
  245. package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
  246. package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
  247. package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
  248. package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
  249. package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
  250. package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
  251. package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
  252. package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
  253. package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
  254. package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
  255. package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
  256. package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
  257. package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
  258. package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
  259. package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
  260. package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
  261. package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
  262. package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
  263. package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
  264. package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
  265. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
  266. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
  267. package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
  268. package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
  269. package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
  270. package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
  271. package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
  272. package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
  273. package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
  274. package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
  275. package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
  276. package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
@@ -7,7 +7,7 @@ import json
7
7
  from pathlib import Path
8
8
 
9
9
  ROOT = Path(__file__).resolve().parent.parent
10
- EXAMPLES_DIR = ROOT / "examples"
10
+ EXAMPLES_DIR = ROOT / "tests" / "fixtures" / "legacy-examples"
11
11
 
12
12
 
13
13
  def enrich_math_project():
@@ -10,10 +10,11 @@ Metrics:
10
10
  - engine_version from engine/versions.py (the version authority)
11
11
  - test_functions grep 'def test_' across tests/
12
12
  - test_files number of collected test modules in tests/
13
- - schema_count schemas/*.json at root + v2/ + v3/ + v4/
13
+ - schema_count schemas/*.json recursively
14
14
  - reference_doc_count references/*.md
15
15
  - gold_annotation_count benchmarks/annotations/gold-Q*.json
16
- - example_packs examples/*/ directories shipping result.json
16
+ - example_packs real examples/*/ directories shipping result.json
17
+ (compatibility symlink aliases are excluded)
17
18
 
18
19
  Usage:
19
20
  python3 scripts/generate_metrics.py # regenerate docs/metrics.json
@@ -56,7 +57,7 @@ def collect() -> dict:
56
57
 
57
58
  example_packs = sorted(
58
59
  p.name for p in (REPO_ROOT / "examples").iterdir()
59
- if p.is_dir() and (p / "result.json").exists()
60
+ if not p.is_symlink() and p.is_dir() and (p / "result.json").exists()
60
61
  ) if (REPO_ROOT / "examples").is_dir() else []
61
62
 
62
63
  return {
@@ -12,7 +12,7 @@ import json
12
12
  from pathlib import Path
13
13
 
14
14
  ROOT = Path(__file__).resolve().parent.parent
15
- EXAMPLES_DIR = ROOT / "examples"
15
+ EXAMPLES_DIR = ROOT / "tests" / "fixtures" / "legacy-examples"
16
16
 
17
17
 
18
18
  def build_math_project():
@@ -10,14 +10,14 @@ stages); the two deterministic stages it executes locally are:
10
10
  adjudicate — Pre-Verdict Gate (scripts/pre_verdict_gate.py) + deterministic
11
11
  confidence (scripts/compute_confidence.py) producing
12
12
  final_verdict.json from raw_verdict.json + evidence.jsonl
13
- present — assemble result.json from the workspace artifacts
13
+ projection — assemble result.json and renderable projections from artifacts
14
14
  (decision = final_verdict.json; claims carry claim_id per
15
15
  report-result.schema.json)
16
16
 
17
17
  Stage machine (execution_plan.json / state.json):
18
18
 
19
19
  frame -> retrieve -> extract -> challenge -> audit -> adjudicate
20
- -> intervene -> evaluate -> present
20
+ -> applicability -> intervene -> evaluate -> projection
21
21
 
22
22
  Each stage writes exactly one primary artifact and is schema-gated against
23
23
  schemas/*. When the artifact is missing the orchestrator either seeds it from
@@ -51,6 +51,9 @@ for _p in (str(ROOT), str(ROOT / "scripts")):
51
51
  if _p not in sys.path:
52
52
  sys.path.insert(0, _p)
53
53
 
54
+ from engine._resources import resource_root # noqa: E402
55
+ ROOT = resource_root()
56
+
54
57
  from run_workspace import (RESOURCE_POLICY_VERSION, STAGES, RunWorkspace, # noqa: E402
55
58
  load_json, load_jsonl, next_run_id, save_jsonl)
56
59
  from pre_verdict_gate import apply_enforcement, evaluate_workspace # noqa: E402
@@ -70,9 +73,10 @@ STAGE_SPEC: dict[str, dict[str, Any]] = {
70
73
  "challenge": {"artifact": "skeptic.json", "schema": None, "jsonl": False, "local": False},
71
74
  "audit": {"artifact": "methodology.json", "schema": "methodology.schema.json", "jsonl": False, "local": False},
72
75
  "adjudicate": {"artifact": "final_verdict.json", "schema": "verdict.schema.json", "jsonl": False, "local": True},
76
+ "applicability": {"artifact": "applicability.json", "schema": None, "jsonl": False, "local": False},
73
77
  "intervene": {"artifact": "intervention.json", "schema": "intervention.schema.json", "jsonl": False, "local": False},
74
78
  "evaluate": {"artifact": "evaluation.json", "schema": "evaluation.schema.json", "jsonl": False, "local": False},
75
- "present": {"artifact": "result.json", "schema": "report-result.schema.json", "jsonl": False, "local": True},
79
+ "projection": {"artifact": "result.json", "schema": "report-result.schema.json", "jsonl": False, "local": True},
76
80
  }
77
81
 
78
82
  #: Phase 33 — canonical failure -> handling-action mapping. Extends the
@@ -143,11 +147,14 @@ _STAGE_BRIEFS: dict[str, str] = {
143
147
  "adjudicate": ("Judge the evidence: write raw_verdict.json (model verdict). The orchestrator "
144
148
  "then runs the Pre-Verdict Gate and deterministic confidence to produce "
145
149
  "final_verdict.json."),
150
+ "applicability": ("Assess whether supported effects apply to the target population, setting, "
151
+ "implementation constraints and outcomes; write applicability.json. Do not "
152
+ "upgrade a decision merely because evidence is present."),
146
153
  "intervene": ("Design the minimal verifiable teaching intervention (phased pilot, "
147
154
  "stop conditions, evidence alignment); write intervention.json."),
148
155
  "evaluate": ("Design the evaluation plan (baseline/post/retention/transfer, task vs learning "
149
156
  "separation); write evaluation.json."),
150
- "present": ("Translate result.json into result.zh.json and render report_spec.json / "
157
+ "projection": ("Translate result.json into result.zh.json and render report_spec.json / "
151
158
  "report.html via the visualization layer."),
152
159
  }
153
160
 
@@ -179,6 +186,7 @@ def init_run(
179
186
  run_id: str | None = None,
180
187
  approve_agent_mcp: bool = False,
181
188
  scp_available: bool | None = None,
189
+ approval_record: dict | None = None,
182
190
  ) -> RunWorkspace:
183
191
  """Create the run workspace + manifest + planning artifacts (Phase 11-13)."""
184
192
  depth = DEPTH_ALIASES.get(depth, depth)
@@ -275,6 +283,16 @@ def init_run(
275
283
  "reason": ("user-approved via --approve-agent-mcp"
276
284
  if approve_agent_mcp else "not yet approved; runs in platform-native mode"),
277
285
  }
286
+ if approval_record:
287
+ agent_mcp_approval["approval_global_path"] = str(_global_approval_path())
288
+ agent_mcp_approval["role_mapping_hash"] = approval_record.get("role_mapping_hash")
289
+ agent_mcp_approval["roles"] = approval_record.get("roles", {})
290
+ agent_mcp_approval["approved_at"] = _utc_now()
291
+ agent_mcp_approval["reason"] = "user-confirmed role mapping (global approval, hash-verified)"
292
+ elif approve_agent_mcp:
293
+ agent_mcp_approval["reason"] = (
294
+ "user-approved via --approve-agent-mcp (boolean only; a role mapping "
295
+ "in the global approval is required before any spawn)")
278
296
 
279
297
  for name, data in (("capability_plan", capability_plan),
280
298
  ("resource_plan", resource_plan),
@@ -306,11 +324,11 @@ def schema_gate(ws: RunWorkspace, stage: str) -> dict[str, Any]:
306
324
  spec = STAGE_SPEC[stage]
307
325
  artifact = spec["artifact"]
308
326
  schema_name = spec["schema"]
309
- if schema_name is None: # challenge: light parseability contract
327
+ if schema_name is None: # lightweight parseability contract
310
328
  data = load_json(ws.path / artifact)
311
329
  ok = bool(data) and isinstance(data, dict)
312
330
  return {"passed": ok, "stage": stage, "artifact": artifact,
313
- "schema": None, "issues": [] if ok else ["skeptic.json missing or unparseable"]}
331
+ "schema": None, "issues": [] if ok else [f"{artifact} missing or unparseable"]}
314
332
 
315
333
  from validate_schema import SchemaError, Validator
316
334
 
@@ -520,14 +538,14 @@ def _assemble_result(ws: RunWorkspace, manifest: dict[str, Any]) -> dict[str, An
520
538
  }
521
539
 
522
540
 
523
- def _run_present(ws: RunWorkspace, manifest: dict[str, Any], question: str,
524
- demo_pack: Path | None = None) -> dict[str, Any]:
525
- """Local present: assemble + validate result.json, seed render artifacts."""
541
+ def _run_projection(ws: RunWorkspace, manifest: dict[str, Any], question: str,
542
+ demo_pack: Path | None = None) -> dict[str, Any]:
543
+ """Build projections after science; this is not a scientific protocol stage."""
526
544
  required = ("final_verdict.json", "intervention.json", "evaluation.json")
527
545
  missing = [name for name in required if not (ws.path / name).is_file()
528
546
  or not load_json(ws.path / name)]
529
547
  if missing:
530
- ws.write_brief("present", question, _STAGE_BRIEFS["present"])
548
+ ws.write_brief("projection", question, _STAGE_BRIEFS["projection"])
531
549
  return {"status": "pending",
532
550
  "detail": f"missing prerequisite artifacts: {', '.join(missing)}"}
533
551
 
@@ -550,7 +568,7 @@ def _run_present(ws: RunWorkspace, manifest: dict[str, Any], question: str,
550
568
  result = _assemble_result(ws, manifest)
551
569
  (ws.path / "result.json").write_text(
552
570
  json.dumps(result, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
553
- gate = schema_gate(ws, "present")
571
+ gate = schema_gate(ws, "projection")
554
572
  if not gate["passed"]:
555
573
  return {"status": "failed", "detail": f"result.json schema gate: {gate['issues']}"}
556
574
  missing_render = [n for n in ("result.zh.json", "report_spec.json", "report.html")
@@ -576,6 +594,10 @@ def run_stage(ws: RunWorkspace, stage: str, *, demo_pack: Path | None = None) ->
576
594
  Deterministic stages execute locally; external stages are either seeded
577
595
  from ``demo_pack`` (demo/test mode) or handed off via a task brief.
578
596
  """
597
+ # Compatibility for callers of the retired name. State/manifests only
598
+ # record ``projection`` from this point forward.
599
+ if stage == "present":
600
+ stage = "projection"
579
601
  ws.trace("stage_started", stage=stage)
580
602
  log.info("stage=%s run=%s start", stage, ws.run_id)
581
603
  spec = STAGE_SPEC[stage]
@@ -598,8 +620,8 @@ def run_stage(ws: RunWorkspace, stage: str, *, demo_pack: Path | None = None) ->
598
620
  # local deterministic stages
599
621
  if stage == "adjudicate":
600
622
  result = _run_adjudicate(ws, question, demo_pack=demo_pack)
601
- elif stage == "present":
602
- result = _run_present(ws, ws.load_manifest(), question, demo_pack=demo_pack)
623
+ elif stage == "projection":
624
+ result = _run_projection(ws, ws.load_manifest(), question, demo_pack=demo_pack)
603
625
  else:
604
626
  # demo/test seeding
605
627
  if demo_pack is not None:
@@ -666,6 +688,21 @@ def _seed_from_demo(ws: RunWorkspace, stage: str, demo_pack: Path) -> dict[str,
666
688
  if (pack / "methodology.json").is_file():
667
689
  (ws.path / "methodology.json").write_bytes((pack / "methodology.json").read_bytes())
668
690
  return {"seeded": True, "detail": "methodology.json seeded from demo pack"}
691
+ elif stage == "applicability":
692
+ if (pack / "applicability.json").is_file():
693
+ (ws.path / "applicability.json").write_bytes((pack / "applicability.json").read_bytes())
694
+ else:
695
+ verdict = load_json(ws.path / "final_verdict.json") or load_json(pack / "verdict.json")
696
+ value = verdict.get("applicability") if isinstance(verdict, dict) else None
697
+ # A demo can only carry the decision's existing applicability
698
+ # boundary; absence remains explicit rather than inferred.
699
+ payload = value if isinstance(value, dict) and value else {
700
+ "status": "NOT_CAPTURED",
701
+ "reason": "demo pack does not provide an applicability assessment",
702
+ }
703
+ (ws.path / "applicability.json").write_text(
704
+ json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
705
+ return {"seeded": True, "detail": "applicability.json seeded from decision boundary (demo)"}
669
706
  elif stage == "intervene":
670
707
  if (pack / "intervention.json").is_file():
671
708
  (ws.path / "intervention.json").write_bytes((pack / "intervention.json").read_bytes())
@@ -732,6 +769,9 @@ def interactive_agent_mcp_setup(approved: bool) -> bool:
732
769
 
733
770
  保留 GitHub 版全部行为(--approve-agent-mcp 旗标、agent_mcp_approval.json、
734
771
  safe_spawn 门);此处只补 run 启动时的交互提示层。非交互终端直接返回原值。
772
+ 新增:Agent MCP 可用时,生成角色→CLI→模型推荐表,询问用户是否采用并
773
+ 固化到全局 ~/.eduevidence/agent_mcp_approval.json(含 hash 防篡改);
774
+ 确认后返回 True,调用方可把该记录带入本次 run 的批准工件。
735
775
  """
736
776
  if approved:
737
777
  return True
@@ -779,13 +819,77 @@ def interactive_agent_mcp_setup(approved: bool) -> bool:
779
819
  return False
780
820
 
781
821
  answer = input("是否启用 Agent MCP 增强模式(推荐)?[Y/n] ").strip().lower()
782
- return answer not in ("n", "no")
822
+ if answer in ("n", "no"):
823
+ return False
824
+
825
+ # Agent MCP 可用:生成推荐表并询问是否固化到全局(简短的 8 角色表)。
826
+ approved_now, _ = _confirm_global_approval()
827
+ return approved_now
828
+
829
+
830
+ def _global_approval_path() -> Path:
831
+ """全局用户批准文件:EDUEVIDENCE_HOME 优先,缺省 ~/.eduevidence。"""
832
+ home = Path(os.environ.get("EDUEVIDENCE_HOME", "~/.eduevidence")).expanduser()
833
+ return home / "agent_mcp_approval.json"
834
+
835
+
836
+ def _confirm_global_approval() -> tuple[bool, dict | None]:
837
+ """构建角色推荐表 → 展示 → 询问 → 固化;返回 (approved, approval_record)。
838
+
839
+ - available CLIs 只扫描本机真实存在的(omp/codex/claude/grok/opencode),
840
+ 不猜模型;无任何可用 CLI 时回退为布尔批准。
841
+ - 任何展示内容都只来自已验证模型清单;推荐行缺少 cli/model 的角色
842
+ 不进映射(safe_spawn 会对该角色保持关闭)。
843
+ - 用户确认后写 ~/.eduevidence/agent_mcp_approval.json(含 role_mapping_hash),
844
+ 后续 run 加载并校验:映射变更即失效,需重新确认。
845
+ """
846
+ import shutil as _shutil
847
+ from integrations.agent_mcp import (build_recommendation_table,
848
+ scan_available_models, write_approval)
849
+
850
+ allowed_clis = [c for c in ("omp", "codex", "claude", "grok", "opencode")
851
+ if _shutil.which(c)]
852
+ if not allowed_clis:
853
+ return True, None
854
+ inventory = scan_available_models(allowed_clis, timeout=20)
855
+ table = build_recommendation_table(allowed_clis, inventory)
856
+ rows = [r for r in table["recommendations"] if r.get("cli") and r.get("model")]
857
+ if not rows:
858
+ print("[startup] 未能从已安装 CLI 解析到已验证模型;按布尔批准启用。")
859
+ return True, None
860
+
861
+ print("[startup] 角色 → CLI / 模型 推荐表(仅基于本机扫描到的可用模型,无固定推荐):")
862
+ for r in rows:
863
+ print(f" {r['role']:<20} → {r['cli']} / {r['model']}")
864
+ summary = table.get("summary", {})
865
+ print(f" cross_model_review(反证异族复核): {summary.get('cross_model_review', 'unknown')}"
866
+ f" | 角色数: {summary.get('role_count', len(rows))}")
867
+ answer = input("采用推荐表并固化到全局批准文件?[Y/n] ").strip().lower()
868
+ if answer in ("n", "no"):
869
+ print("[startup] 未固化;本次以平台原生模式运行(可用 --approve-agent-mcp 跳过询问)。")
870
+ return False, None
871
+ roles = {r["role"]: {"cli": r["cli"], "model": r["model"]} for r in rows}
872
+ path = _global_approval_path()
873
+ path.parent.mkdir(parents=True, exist_ok=True)
874
+ record = write_approval(path, roles, sorted(allowed_clis))
875
+ print(f"[startup] 已固化 → {path}")
876
+ return True, record
877
+
878
+
879
+ def load_global_approval() -> dict | None:
880
+ """加载全局批准文件(missing/corrupt -> None;有效期交由
881
+ integrations.agent_mcp.is_approval_current 判定)。"""
882
+ from integrations.agent_mcp import load_approval
883
+ return load_approval(_global_approval_path())
783
884
 
784
885
 
785
886
  def _cmd_run(args: argparse.Namespace) -> int:
786
887
  approve = args.approve_agent_mcp or interactive_agent_mcp_setup(args.approve_agent_mcp)
888
+ approval_record = None
889
+ if approve:
890
+ approval_record = load_global_approval()
787
891
  ws = init_run(Path(args.runs_dir), args.question, depth=args.depth, run_id=args.run_id,
788
- approve_agent_mcp=approve)
892
+ approve_agent_mcp=approve, approval_record=approval_record)
789
893
  print(f"workspace created: {ws.path}")
790
894
  print(f"manifest: {json.dumps(ws.load_manifest(), ensure_ascii=False, indent=2)}")
791
895
  if args.dry_run:
@@ -984,9 +1088,15 @@ def _cmd_synthesize(args) -> int:
984
1088
  def _cmd_benchmark(args) -> int:
985
1089
  import benchmark_v3 as bv3
986
1090
  if args.action == "run":
987
- return bv3.main(["run", "--baselines", args.baselines, "--questions", args.questions,
988
- "--repeats", str(args.repeats), "--driver", args.driver,
989
- "--out", args.out, "--budget-tokens", str(args.budget)])
1091
+ argv = ["run", "--baselines", args.baselines, "--questions", args.questions,
1092
+ "--repeats", str(args.repeats), "--out", args.out,
1093
+ "--budget-tokens", str(args.budget), "--model", args.model,
1094
+ "--thinking", args.thinking]
1095
+ if getattr(args, "ids", None):
1096
+ argv.extend(["--ids", args.ids])
1097
+ if args.driver:
1098
+ argv.extend(["--driver", args.driver])
1099
+ return bv3.main(argv)
990
1100
  if args.action == "eval":
991
1101
  return bv3.main(["eval", "--run", args.run, "--annotations", args.annotations])
992
1102
  if args.action == "report":
@@ -1172,6 +1282,17 @@ def _cmd_report(args) -> int:
1172
1282
  return 0
1173
1283
 
1174
1284
 
1285
+ def _cmd_export(args) -> int:
1286
+ from engine.judge_pack import export_judge_pack
1287
+ from engine.project import ProjectWorkspace
1288
+ project = ProjectWorkspace.open(_home(args), args.project)
1289
+ output = Path(args.out) if args.out else project.path / "exports" / "judge-pack"
1290
+ manifest = export_judge_pack(project, output)
1291
+ print(json.dumps({"output": str(output), "files": len(manifest["copied_files"]),
1292
+ "missing_categories": manifest["missing_categories"]}, ensure_ascii=False))
1293
+ return 0
1294
+
1295
+
1175
1296
  def _cmd_migrate(args) -> int:
1176
1297
  from engine.migration import migrate_v1_pack
1177
1298
  result = migrate_v1_pack(args.pack, home=_home(args), title=args.title)
@@ -1194,6 +1315,17 @@ def _cmd_search(args) -> int:
1194
1315
  return 0
1195
1316
 
1196
1317
 
1318
+ def _cmd_search_plan(args) -> int:
1319
+ from search_provenance import main as search_plan_main
1320
+ argv = [args.query, "--out", str(args.out), "--domain", args.domain,
1321
+ "--limit", str(args.limit), "--channel", args.channel, "--policy", args.policy]
1322
+ for concept in args.concept:
1323
+ argv.extend(["--concept", concept])
1324
+ for synonym in args.synonym:
1325
+ argv.extend(["--synonym", synonym])
1326
+ return search_plan_main(argv)
1327
+
1328
+
1197
1329
  def _cmd_did(args) -> int:
1198
1330
  from did_regression import run_did_analysis
1199
1331
  res = run_did_analysis(str(args.csv))
@@ -1329,6 +1461,13 @@ def main(argv: list[str] | None = None) -> int:
1329
1461
  p_report.add_argument("--home", default=None)
1330
1462
  p_report.set_defaults(func=_cmd_report)
1331
1463
 
1464
+ p_export = sub.add_parser("export", help="export a project evidence pack")
1465
+ p_export.add_argument("kind", choices=["judge-pack"])
1466
+ p_export.add_argument("project", help="project id")
1467
+ p_export.add_argument("--home", default=None)
1468
+ p_export.add_argument("--out", default=None)
1469
+ p_export.set_defaults(func=_cmd_export)
1470
+
1332
1471
 
1333
1472
  p_pilot = sub.add_parser("pilot", help="V3 Decision-to-Outcome Loop")
1334
1473
  p_pilot.add_argument("action", choices=["register", "import", "analyze-link", "redecide"])
@@ -1363,11 +1502,15 @@ def main(argv: list[str] | None = None) -> int:
1363
1502
  p_bench.add_argument("action", choices=["run", "eval", "report"])
1364
1503
  p_bench.add_argument("--baselines", default="B2_standard_agent,B3_eduevidence_single")
1365
1504
  p_bench.add_argument("--questions", default="benchmarks/questions.jsonl")
1505
+ p_bench.add_argument("--ids", default=None, help="comma-separated question ids to run")
1366
1506
  p_bench.add_argument("--repeats", type=int, default=3)
1367
1507
  p_bench.add_argument("--driver", default=None, choices=["api", "cli", "sim"],
1368
1508
  help="api | cli (omp) | sim (harness validation only); default: auto (api > cli > sim)")
1369
1509
  p_bench.add_argument("--out", default="benchmarks/empirical/run-001")
1370
1510
  p_bench.add_argument("--budget", type=int, default=1000000)
1511
+ p_bench.add_argument("--model", default="",
1512
+ help="model for --driver cli (required; no unconfirmed default)")
1513
+ p_bench.add_argument("--thinking", default="max", choices=["low", "high", "max"])
1371
1514
  p_bench.add_argument("--run", default=None, help="run dir (eval/report)")
1372
1515
  p_bench.add_argument("--annotations", default="benchmarks/annotations")
1373
1516
  p_bench.add_argument("--report", default="benchmarks/empirical/v3-report.md")
@@ -1414,6 +1557,17 @@ def main(argv: list[str] | None = None) -> int:
1414
1557
  p_srch.add_argument("--academic", action="store_true", help="academic only")
1415
1558
  p_srch.set_defaults(func=_cmd_search)
1416
1559
 
1560
+ p_sp = sub.add_parser("search-plan", help="audited, bounded search with provenance export")
1561
+ p_sp.add_argument("query", help="research question")
1562
+ p_sp.add_argument("--out", required=True, type=Path)
1563
+ p_sp.add_argument("--domain", default="education", choices=["education", "policy"])
1564
+ p_sp.add_argument("--concept", action="append", default=[])
1565
+ p_sp.add_argument("--synonym", action="append", default=[])
1566
+ p_sp.add_argument("--limit", type=int, default=10)
1567
+ p_sp.add_argument("--channel", default="all", choices=["all", "academic", "web"])
1568
+ p_sp.add_argument("--policy", default="2026.09")
1569
+ p_sp.set_defaults(func=_cmd_search_plan)
1570
+
1417
1571
  p_did = sub.add_parser("did", help="run DID regression on classroom CSV")
1418
1572
  p_did.add_argument("csv", type=Path, help="CSV file path")
1419
1573
  p_did.set_defaults(func=_cmd_did)
@@ -10,8 +10,7 @@ from pathlib import Path
10
10
  ROOT = Path(__file__).resolve().parent.parent
11
11
 
12
12
  PROJECTS = [
13
- "examples/highschool-math-ai-tutor",
14
- "examples/esl-academic-writing-ai",
13
+ "examples/workplace-ai-assistant",
15
14
  "examples/ai-coding-assistant-evidence"
16
15
  ]
17
16