eduevidence 5.2.0 → 6.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (276) hide show
  1. package/README.md +54 -42
  2. package/README.zh-CN.md +51 -28
  3. package/SKILL.md +390 -133
  4. package/agents/openai.yaml +4 -0
  5. package/assets/readme/controlled-execution.svg +34 -0
  6. package/assets/readme/logo.png +0 -0
  7. package/assets/readme/research-workflow.svg +56 -0
  8. package/assets/readme/studio-graph.png +0 -0
  9. package/assets/readme/studio-overview.png +0 -0
  10. package/assets/readme/studio-reports.png +0 -0
  11. package/autoevolve/config.yaml +17 -0
  12. package/autoevolve/program.md +25 -0
  13. package/autoevolve/protected.manifest.yaml +34 -0
  14. package/benchmarks/adversarial/cases.jsonl +7 -0
  15. package/benchmarks/evidence-library.json +5268 -0
  16. package/benchmarks/partitions.json +8 -0
  17. package/docs/architecture.md +220 -0
  18. package/docs/autoresearch-evolution-plan.md +2903 -0
  19. package/docs/autoresearch-implementation-status.md +101 -0
  20. package/docs/demo-storyboard.md +20 -0
  21. package/docs/demo-workplace-ai.md +92 -0
  22. package/docs/demo.md +32 -0
  23. package/docs/install-guide.md +150 -0
  24. package/docs/orchestration-role-model.md +1254 -0
  25. package/docs/release-closeout/README.md +17 -0
  26. package/docs/release-closeout/frontend-acceptance.md +23 -0
  27. package/docs/release-closeout/issues.md +19 -0
  28. package/docs/release-closeout/verification.md +28 -0
  29. package/docs/release-contract.md +108 -0
  30. package/docs/research-studio-guide.zh-CN.md +166 -0
  31. package/eduevidence_cli.py +17 -11
  32. package/engine/_resources.py +13 -0
  33. package/engine/autoevolve/__init__.py +3 -0
  34. package/engine/autoevolve/agent_view.py +167 -0
  35. package/engine/autoevolve/core.py +357 -0
  36. package/engine/autoevolve/events.py +11 -0
  37. package/engine/autoevolve/git_workspace.py +77 -0
  38. package/engine/autoevolve/projection.py +23 -0
  39. package/engine/autoevolve/runner.py +413 -0
  40. package/engine/autoevolve/trust.py +146 -0
  41. package/engine/autoresearch/__init__.py +6 -0
  42. package/engine/autoresearch/commit.py +132 -0
  43. package/engine/autoresearch/contracts.py +126 -0
  44. package/engine/autoresearch/controller.py +207 -0
  45. package/engine/autoresearch/events.py +12 -0
  46. package/engine/autoresearch/gap_priority.py +168 -0
  47. package/engine/autoresearch/projection.py +30 -0
  48. package/engine/autoresearch/research_memory.py +59 -0
  49. package/engine/autoresearch/saturation.py +91 -0
  50. package/engine/briefs.py +2 -1
  51. package/engine/capabilities.py +1 -0
  52. package/engine/contracts.py +3 -1
  53. package/engine/evidencecore.py +7 -5
  54. package/engine/gaps.py +90 -51
  55. package/engine/judge_pack.py +65 -0
  56. package/engine/library_builtin.py +3 -1
  57. package/engine/living.py +2 -1
  58. package/engine/meta_synthesis.py +3 -1
  59. package/engine/orchestration.py +460 -0
  60. package/engine/pilot.py +2 -1
  61. package/engine/project.py +2 -2
  62. package/engine/research_service.py +113 -0
  63. package/engine/studio_read_model.py +400 -0
  64. package/engine/tribunal.py +1 -2
  65. package/engine/update.py +1 -0
  66. package/engine/versions.py +1 -1
  67. package/engine/worker_result.py +109 -0
  68. package/engine/workflows.py +70 -0
  69. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1720 -0
  70. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
  71. package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
  72. package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
  73. package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
  74. package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
  75. package/examples/ai-coding-assistant-evidence/frame.json +48 -0
  76. package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
  77. package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
  78. package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
  79. package/examples/ai-coding-assistant-evidence/report_spec.json +219 -0
  80. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2614 -0
  81. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2614 -0
  82. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2614 -0
  83. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2614 -0
  84. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2614 -0
  85. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1720 -0
  86. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1720 -0
  87. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1720 -0
  88. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1720 -0
  89. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1720 -0
  90. package/examples/ai-coding-assistant-evidence/result.json +1453 -0
  91. package/examples/ai-coding-assistant-evidence/result.zh.json +1453 -0
  92. package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
  93. package/examples/ai-coding-assistant-evidence/verdict.json +103 -0
  94. package/examples/workplace-ai-assistant/claims.jsonl +4 -0
  95. package/examples/workplace-ai-assistant/evaluation.json +19 -0
  96. package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
  97. package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
  98. package/examples/workplace-ai-assistant/frame.json +41 -0
  99. package/examples/workplace-ai-assistant/intervention.json +27 -0
  100. package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
  101. package/examples/workplace-ai-assistant/methodology.json +60 -0
  102. package/examples/workplace-ai-assistant/report_spec.json +55 -0
  103. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2484 -0
  104. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2484 -0
  105. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2484 -0
  106. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2484 -0
  107. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2484 -0
  108. package/examples/workplace-ai-assistant/result.json +553 -0
  109. package/examples/workplace-ai-assistant/result.zh.json +553 -0
  110. package/examples/workplace-ai-assistant/search_log.json +19 -0
  111. package/examples/workplace-ai-assistant/sources.jsonl +3 -0
  112. package/examples/workplace-ai-assistant/validation_result.json +9 -0
  113. package/examples/workplace-ai-assistant/verdict.json +52 -0
  114. package/install.sh +7 -7
  115. package/integrations/orchestration_dispatch.py +146 -0
  116. package/package.json +37 -3
  117. package/pyproject.toml +11 -20
  118. package/references/autoresearch.md +30 -0
  119. package/references/evaluation-policy.md +24 -0
  120. package/references/orchestration.md +22 -0
  121. package/references/scientific-invariants.md +19 -0
  122. package/retrieval/audit.py +154 -0
  123. package/schemas/intervention.schema.json +106 -21
  124. package/schemas/report-result.schema.json +9 -1
  125. package/schemas/v2/project.schema.json +2 -2
  126. package/schemas/v2/run.schema.json +1 -1
  127. package/schemas/vNext/autoevolve-session.schema.json +1 -0
  128. package/schemas/vNext/eval-snapshot.schema.json +1 -0
  129. package/schemas/vNext/execution-plan.schema.json +1 -0
  130. package/schemas/vNext/gap-priority.schema.json +1 -0
  131. package/schemas/vNext/negative-search-record.schema.json +1 -0
  132. package/schemas/vNext/research-iteration.schema.json +1 -0
  133. package/schemas/vNext/research-strategy.schema.json +1 -0
  134. package/schemas/vNext/skill-experiment.schema.json +1 -0
  135. package/schemas/vNext/task-spec.schema.json +1 -0
  136. package/schemas/vNext/worker-result.schema.json +1 -0
  137. package/scripts/benchmark_judge.py +2 -2
  138. package/scripts/benchmark_v3.py +26 -43
  139. package/scripts/build_esl_artifacts.py +2 -2
  140. package/scripts/build_evidence_library.py +2 -2
  141. package/scripts/build_gh_pages.py +98 -0
  142. package/scripts/build_readme_diagrams.py +72 -0
  143. package/scripts/build_report_variants.py +85 -0
  144. package/scripts/check_autoresearch_invariants.py +95 -0
  145. package/scripts/daily_evolve.py +30 -0
  146. package/scripts/dashboard_server.py +130 -101
  147. package/scripts/did_regression.py +5 -30
  148. package/scripts/enrich_projects_human_and_lieflat.py +1 -1
  149. package/scripts/generate_metrics.py +4 -3
  150. package/scripts/generate_new_projects.py +1 -1
  151. package/scripts/orchestrator.py +172 -18
  152. package/scripts/rebake_all_5themes.py +1 -2
  153. package/scripts/research_auto_cli.py +475 -0
  154. package/scripts/run_workspace.py +17 -7
  155. package/scripts/search_provenance.py +64 -0
  156. package/scripts/serve_web.py +9 -10
  157. package/scripts/skill_lint.py +1 -1
  158. package/scripts/skill_payload.py +78 -0
  159. package/scripts/validate_schema.py +15 -1
  160. package/scripts/vnext_cli.py +133 -0
  161. package/setup.py +12 -0
  162. package/skill/roles/registry.yaml +45 -0
  163. package/skill/sub-skills/report-generation/SKILL.md +12 -6
  164. package/skill/task-briefs/applicability.md +3 -0
  165. package/skill/task-briefs/projection.md +3 -0
  166. package/skill/workflows/decision-and-pilot.md +10 -0
  167. package/skill/workflows/evaluate-and-update.md +10 -0
  168. package/skill/workflows/evidence-review.md +13 -0
  169. package/visualization/eduevidence-report/assets/base.css +2 -2
  170. package/visualization/eduevidence-report/assets/reader.css +752 -0
  171. package/visualization/eduevidence-report/assets/reader.js +132 -0
  172. package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
  173. package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
  174. package/visualization/eduevidence-report/scripts/build_report.py +58 -65
  175. package/visualization/eduevidence-report/scripts/lieflat_engine.py +24 -100
  176. package/visualization/eduevidence-report/themes/academic.css +1 -1
  177. package/visualization/eduevidence-report/themes/claude.css +1 -1
  178. package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
  179. package/visualization/eduevidence-report/themes/datalab.css +2 -2
  180. package/visualization/eduevidence-report/themes/presentation.css +2 -2
  181. package/web/README.md +18 -0
  182. package/web/index.html +53 -0
  183. package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
  184. package/web/studio/assets/index-CzXocaGv.css +1 -0
  185. package/web/studio/assets/index-pa7jD7n4.js +230 -0
  186. package/web/studio/config.json +1 -0
  187. package/web/studio/index.html +14 -0
  188. package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
  189. package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
  190. package/engine/__pycache__/bias.cpython-312.pyc +0 -0
  191. package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
  192. package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
  193. package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
  194. package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
  195. package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
  196. package/engine/__pycache__/events.cpython-312.pyc +0 -0
  197. package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
  198. package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
  199. package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
  200. package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
  201. package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
  202. package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
  203. package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
  204. package/engine/__pycache__/ids.cpython-312.pyc +0 -0
  205. package/engine/__pycache__/library.cpython-312.pyc +0 -0
  206. package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
  207. package/engine/__pycache__/living.cpython-312.pyc +0 -0
  208. package/engine/__pycache__/log.cpython-312.pyc +0 -0
  209. package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
  210. package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
  211. package/engine/__pycache__/migration.cpython-312.pyc +0 -0
  212. package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
  213. package/engine/__pycache__/paths.cpython-312.pyc +0 -0
  214. package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
  215. package/engine/__pycache__/planner.cpython-312.pyc +0 -0
  216. package/engine/__pycache__/project.cpython-312.pyc +0 -0
  217. package/engine/__pycache__/projections.cpython-312.pyc +0 -0
  218. package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
  219. package/engine/__pycache__/run.cpython-312.pyc +0 -0
  220. package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
  221. package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
  222. package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
  223. package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
  224. package/engine/__pycache__/update.cpython-312.pyc +0 -0
  225. package/engine/__pycache__/versions.cpython-312.pyc +0 -0
  226. package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
  227. package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
  228. package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
  229. package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
  230. package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
  231. package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
  232. package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
  233. package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
  234. package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
  235. package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
  236. package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
  237. package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
  238. package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
  239. package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
  240. package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
  241. package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
  242. package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
  243. package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
  244. package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
  245. package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
  246. package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
  247. package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
  248. package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
  249. package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
  250. package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
  251. package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
  252. package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
  253. package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
  254. package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
  255. package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
  256. package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
  257. package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
  258. package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
  259. package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
  260. package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
  261. package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
  262. package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
  263. package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
  264. package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
  265. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
  266. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
  267. package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
  268. package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
  269. package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
  270. package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
  271. package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
  272. package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
  273. package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
  274. package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
  275. package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
  276. package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
@@ -0,0 +1,59 @@
1
+ from __future__ import annotations
2
+ import json
3
+ from pathlib import Path
4
+ from typing import Any
5
+ from .contracts import NegativeSearchRecord, ResearchIteration
6
+
7
+
8
+ class ResearchMemory:
9
+ def __init__(self, root: str | Path):
10
+ self.root = Path(root)
11
+ self.root.mkdir(parents=True, exist_ok=True)
12
+ self.iterations_path = self.root / "research-iterations.jsonl"
13
+ self.negative_path = self.root / "negative-searches.jsonl"
14
+
15
+ @staticmethod
16
+ def _append(path: Path, record: dict[str, Any]) -> None:
17
+ with path.open("a", encoding="utf-8") as f:
18
+ f.write(json.dumps(record, ensure_ascii=False, sort_keys=True) + "\n")
19
+
20
+ def append_iteration(self, iteration: ResearchIteration) -> None:
21
+ iteration.validate()
22
+ self._append(self.iterations_path, iteration.as_dict())
23
+
24
+ def append_negative_search(self, record: NegativeSearchRecord) -> None:
25
+ record.validate()
26
+ self._append(self.negative_path, record.__dict__)
27
+
28
+ def load_iterations(
29
+ self,
30
+ gap_id: str | None = None,
31
+ *,
32
+ gap_lineage_key: str | None = None,
33
+ ) -> list[dict[str, Any]]:
34
+ """Load iteration history, preferring stable lineage across revisions.
35
+
36
+ Legacy rows without `gap_lineage_key` remain queryable by `gap_id`.
37
+ When a lineage key is supplied, new keyed rows match by lineage and old
38
+ unkeyed rows may additionally match the supplied gap_id for migration.
39
+ """
40
+ if not self.iterations_path.exists():
41
+ return []
42
+ rows = [
43
+ json.loads(line)
44
+ for line in self.iterations_path.read_text(encoding="utf-8").splitlines()
45
+ if line.strip()
46
+ ]
47
+ if gap_lineage_key is not None:
48
+ return [
49
+ row for row in rows
50
+ if row.get("gap_lineage_key") == gap_lineage_key
51
+ or (
52
+ not row.get("gap_lineage_key")
53
+ and gap_id is not None
54
+ and row.get("gap_id") == gap_id
55
+ )
56
+ ]
57
+ if gap_id is not None:
58
+ return [row for row in rows if row.get("gap_id") == gap_id]
59
+ return rows
@@ -0,0 +1,91 @@
1
+ from __future__ import annotations
2
+ from dataclasses import dataclass
3
+ from typing import Any
4
+
5
+
6
+ @dataclass(frozen=True)
7
+ class SaturationResult:
8
+ saturated: bool
9
+ low_yield_streak: int
10
+ strategy_diversity_exhausted: bool
11
+ rationale: tuple[str, ...]
12
+
13
+
14
+ def _is_low_yield(row: dict[str, Any]) -> bool:
15
+ gain = row.get("evidence_gain") or {}
16
+ unique = int(gain.get("unique_eligible_evidence", 0) or 0)
17
+ direct = int(gain.get("direct_outcome_findings", 0) or 0)
18
+ delta = float(gain.get("decision_boundary_delta", 0) or 0)
19
+ duplicate_rate = float(gain.get("duplicate_rate", 0) or 0)
20
+ candidate_sources = row.get("candidate_sources") or []
21
+ no_candidates = len(candidate_sources) == 0
22
+ return (
23
+ unique == 0
24
+ and direct == 0
25
+ and abs(delta) < 1e-12
26
+ and (duplicate_rate >= 0.5 or no_candidates)
27
+ )
28
+
29
+
30
+ def detect_saturation(
31
+ iterations: list[dict[str, Any]],
32
+ *,
33
+ min_consecutive: int = 2,
34
+ available_strategy_types: set[str] | None = None,
35
+ ) -> SaturationResult:
36
+ """Detect bounded secondary-search saturation.
37
+
38
+ Strategy diversity is computed across the full history for the gap, while
39
+ the low-yield condition is intentionally a trailing streak. Empty searches
40
+ count as low-yield even when duplicate_rate is zero; otherwise a provider
41
+ returning no candidates could keep the loop alive forever.
42
+ """
43
+ attempted_all = {
44
+ str((row.get("strategy") or {}).get("experiment_type", ""))
45
+ for row in iterations
46
+ if str((row.get("strategy") or {}).get("experiment_type", ""))
47
+ }
48
+ streak = 0
49
+ for row in reversed(iterations):
50
+ if _is_low_yield(row):
51
+ streak += 1
52
+ else:
53
+ break
54
+
55
+ if available_strategy_types:
56
+ diversity_exhausted = available_strategy_types.issubset(attempted_all)
57
+ else:
58
+ diversity_exhausted = len(attempted_all) >= 2
59
+
60
+ rationale: list[str] = []
61
+ if streak >= min_consecutive:
62
+ rationale.append(
63
+ f"{streak} consecutive iterations produced no unique/direct evidence or decision-boundary change"
64
+ )
65
+ if diversity_exhausted:
66
+ rationale.append("strategy diversity exhausted for the configured search space")
67
+ return SaturationResult(
68
+ streak >= min_consecutive and diversity_exhausted,
69
+ streak,
70
+ diversity_exhausted,
71
+ tuple(rationale),
72
+ )
73
+
74
+
75
+ def transition_to_empirical(
76
+ *,
77
+ dvi_band: str,
78
+ decision_material: bool,
79
+ unresolved: bool,
80
+ saturation: SaturationResult,
81
+ ethics_feasible: bool,
82
+ ) -> tuple[bool, tuple[str, ...]]:
83
+ checks = [
84
+ (dvi_band.upper() == "HIGH", "gap DVI is HIGH"),
85
+ (decision_material, "gap is material to the decision"),
86
+ (unresolved, "gap remains unresolved"),
87
+ (saturation.saturated, "secondary search is saturated"),
88
+ (ethics_feasible, "empirical study is ethically/operationally feasible"),
89
+ ]
90
+ reasons = tuple(text for ok, text in checks if ok)
91
+ return all(ok for ok, _ in checks), reasons
package/engine/briefs.py CHANGED
@@ -14,6 +14,7 @@ from pathlib import Path
14
14
 
15
15
  from engine.contracts import load_schema, schema_path
16
16
  from engine.planner import PlanStep
17
+ from engine.project import ProjectWorkspace
17
18
 
18
19
 
19
20
  def _schema_section(schema: dict) -> str:
@@ -73,7 +74,7 @@ def build_task_brief(step: PlanStep, *, project: ProjectWorkspace,
73
74
  f"Validate the output with `engine.contracts.validate_record({schema_name!r}, record)` — "
74
75
  f"it must return [] (empty errors).",
75
76
  "",
76
- f"## Output path",
77
+ "## Output path",
77
78
  str(output_path),
78
79
  "",
79
80
  "## Inputs",
@@ -18,6 +18,7 @@ class CapabilitySpec:
18
18
  output_contracts: tuple[str, ...]
19
19
  deterministic_local: bool
20
20
  scientific_gate: str | None
21
+ workflow_ids: tuple[str, ...] = ()
21
22
 
22
23
 
23
24
  _REGISTRY: dict[str, CapabilitySpec] = {}
@@ -12,7 +12,9 @@ from typing import Callable
12
12
 
13
13
  from scripts.validate_schema import SchemaError, validate
14
14
 
15
- _REPO_SCHEMA_DIR = Path(__file__).resolve().parent.parent / "schemas" / "v2"
15
+ from engine._resources import resource_root
16
+
17
+ _REPO_SCHEMA_DIR = resource_root() / "schemas" / "v2"
16
18
 
17
19
 
18
20
  def _resolve_schema_dir() -> Path:
@@ -12,8 +12,7 @@ v4 领域包机制:domains/ 注册表 + 领域契约加载 + frame 校验。
12
12
  education 域只是"指向现有契约"的注册:不新增任何逻辑路径、不引入新 schema
13
13
  或新校验器。领域选择(domain select)由主 agent 接 CLI 完成,引擎层不做选择。
14
14
 
15
- 路径解析:当前按仓库布局(domains/ 在仓库根目录)解析;wheel 安装场景的
16
- share/ 回退留给后续步骤(pyproject data-files 未包含 domains/)。
15
+ 路径解析:支持仓库、独立 Skill 与 wheel 的 share/eduevidence 资源布局。
17
16
  """
18
17
 
19
18
  from __future__ import annotations
@@ -22,7 +21,9 @@ import json
22
21
  from pathlib import Path
23
22
  from typing import Any
24
23
 
25
- REPO_ROOT = Path(__file__).resolve().parent.parent
24
+ from engine._resources import resource_root
25
+
26
+ REPO_ROOT = resource_root()
26
27
 
27
28
 
28
29
  def _resolve_domains_dir() -> Path:
@@ -114,7 +115,7 @@ def _validate_contracts(entry: dict) -> None:
114
115
 
115
116
  - frame_schema / outcome_taxonomy / methodology_checklist:文件存在且
116
117
  为可解析 JSON(指针引用另校验指针内容);
117
- - golds_dir / references_dir:目录存在(null 视为"无此契约",跳过)。
118
+ - references_dir:目录存在;golds_dir 属于独立 evaluator 资源,不是研究运行依赖。
118
119
  """
119
120
  domain_id = entry["id"]
120
121
 
@@ -142,7 +143,8 @@ def _validate_contracts(entry: dict) -> None:
142
143
  check_file("frame_schema")
143
144
  check_file("outcome_taxonomy")
144
145
  check_file("methodology_checklist")
145
- check_dir("golds_dir")
146
+ # Evaluation annotations (including holdout answers) are intentionally absent
147
+ # from shipped Skills. Benchmark consumers validate their own input corpus.
146
148
  check_dir("references_dir")
147
149
 
148
150
 
package/engine/gaps.py CHANGED
@@ -4,10 +4,15 @@ A KnowledgeGap is not free-form "future work": it is derived from coverage —
4
4
  the research frame's requested outcomes vs what the graph's Findings
5
5
  actually measure. A task-performance Finding never covers a retention or
6
6
  transfer gap.
7
+
8
+ `gap_id` identifies one revision-local gap artifact. `extensions.autoresearch_key`
9
+ is a stable semantic lineage key so bounded research memory survives graph
10
+ revisions when the same unresolved gap is re-derived.
7
11
  """
8
12
 
9
13
  from __future__ import annotations
10
14
 
15
+ import hashlib
11
16
  import json
12
17
  from pathlib import Path
13
18
 
@@ -16,23 +21,35 @@ from engine.graph_store import GraphStore
16
21
  from engine.ids import new_local_id
17
22
  from engine.synthesis import ClaimSynthesis
18
23
 
19
- # outcome_type names for timepoint-like gaps (frame.requested_outcomes entries
20
- # may carry outcome_type or be plain strings; we match on outcome_type)
21
24
  _RETENTION_TYPES = {"retention", "long_term", "learning_retention"}
22
25
  _TRANSFER_TYPES = {"transfer", "transfer_learning", "far_transfer"}
23
26
  _TASK_PERFORMANCE = {"task_performance", "assignment_score", "task_completion"}
24
27
  _LEARNING = {"learning"}
25
28
 
26
29
 
30
+ def _autoresearch_key(
31
+ gap_type: str,
32
+ *,
33
+ related_claims: list[str],
34
+ related_outcomes: list[str],
35
+ semantic_token: str,
36
+ ) -> str:
37
+ payload = {
38
+ "gap_type": gap_type,
39
+ "related_claims": sorted(related_claims),
40
+ "related_outcomes": sorted(related_outcomes),
41
+ "semantic_token": semantic_token.strip().lower(),
42
+ }
43
+ digest = hashlib.sha256(
44
+ json.dumps(payload, sort_keys=True, ensure_ascii=False).encode("utf-8")
45
+ ).hexdigest()[:20]
46
+ return f"KGK-{digest}"
47
+
48
+
27
49
  def derive_gaps(*, store: GraphStore,
28
50
  syntheses: tuple[ClaimSynthesis, ...] | None = None,
29
51
  frame: dict | None = None) -> list[dict]:
30
- """Derive structured gaps from graph coverage vs the research frame.
31
-
32
- `frame` carries `requested_outcomes` (list of outcome names/types) and
33
- optionally `target_population`. Findings' outcome types come from the
34
- graph's outcomes table.
35
- """
52
+ """Derive structured gaps from graph coverage vs the research frame."""
36
53
  frame = frame or {}
37
54
  requested = frame.get("requested_outcomes") or []
38
55
  if not requested and frame.get("target_outcomes"):
@@ -47,30 +64,34 @@ def derive_gaps(*, store: GraphStore,
47
64
  covered_types.add(o.get("outcome_type", ""))
48
65
 
49
66
  claims = store.read_table("claims")
50
- claim_ids = [c["claim_id"] for c in claims]
51
-
52
67
  gaps: list[dict] = []
53
68
  rev = store.active_revision()
54
69
 
55
70
  def add(gap_type: str, priority: str, reasoning: str,
56
71
  related_claims: list[str] | None = None,
57
- related_outcomes: list[str] | None = None):
72
+ related_outcomes: list[str] | None = None,
73
+ semantic_token: str = ""):
74
+ related_claims = related_claims or []
75
+ related_outcomes = related_outcomes or []
76
+ key = _autoresearch_key(
77
+ gap_type,
78
+ related_claims=related_claims,
79
+ related_outcomes=related_outcomes,
80
+ semantic_token=semantic_token or reasoning,
81
+ )
58
82
  gaps.append({
59
83
  "gap_id": new_local_id("GAP", {g["gap_id"] for g in gaps}),
60
84
  "gap_type": gap_type,
61
- "related_claim_ids": related_claims or [],
62
- "related_outcome_ids": related_outcomes or [],
85
+ "related_claim_ids": related_claims,
86
+ "related_outcome_ids": related_outcomes,
63
87
  "priority": priority,
64
88
  "reasoning": reasoning,
65
89
  "status": "open",
66
90
  "derived_from_graph_revision": rev,
67
- "extensions": {},
91
+ "extensions": {"autoresearch_key": key},
68
92
  })
69
93
 
70
94
  def _req_kind(req) -> tuple[str, str]:
71
- """Classify a requested outcome: retention | transfer |
72
- task_performance | learning | other. Type-aware: names are matched
73
- only within the outcome's declared type, never type-blind."""
74
95
  if isinstance(req, dict):
75
96
  req_name = str(req.get("name", "")).lower()
76
97
  req_type = str(req.get("outcome_type", "")).lower()
@@ -86,20 +107,17 @@ def derive_gaps(*, store: GraphStore,
86
107
  return "learning", req.get("name", "") if isinstance(req, dict) else str(req)
87
108
  return "other", req.get("name", "") if isinstance(req, dict) else str(req)
88
109
 
89
- _RETENTION_COVER = _RETENTION_TYPES
90
- _TRANSFER_COVER = _TRANSFER_TYPES
91
110
  def covered_for_kind(kind: str) -> bool:
92
111
  if kind == "retention":
93
- return bool(covered_types & _RETENTION_COVER)
112
+ return bool(covered_types & _RETENTION_TYPES)
94
113
  if kind == "transfer":
95
- return bool(covered_types & _TRANSFER_COVER)
114
+ return bool(covered_types & _TRANSFER_TYPES)
96
115
  if kind == "task_performance":
97
116
  return bool(covered_types & _TASK_PERFORMANCE)
98
117
  if kind == "learning":
99
118
  return bool(covered_types & _LEARNING)
100
119
  return False
101
120
 
102
- # one pass per requested outcome; each gap emitted exactly once
103
121
  seen: set[tuple[str, str]] = set()
104
122
  for req in requested:
105
123
  kind, label = _req_kind(req)
@@ -112,49 +130,70 @@ def derive_gaps(*, store: GraphStore,
112
130
  if covered_for_kind(kind):
113
131
  continue
114
132
  if kind == "retention":
115
- add("missing_retention", "high",
133
+ add(
134
+ "missing_retention", "high",
116
135
  f"frame requests retention outcome {label!r} but the graph has "
117
- f"no retention-type measurement; task-performance coverage does "
118
- f"not count (RULE 3)")
136
+ "no retention-type measurement; task-performance coverage does "
137
+ "not count (RULE 3)",
138
+ semantic_token=f"requested_outcome:{label}",
139
+ )
119
140
  elif kind == "transfer":
120
- add("missing_transfer", "high",
141
+ add(
142
+ "missing_transfer", "high",
121
143
  f"frame requests transfer outcome {label!r} but the graph has "
122
- f"no transfer-type measurement; AI-assisted task performance "
123
- f"does not count (RULE 3)")
144
+ "no transfer-type measurement; AI-assisted task performance "
145
+ "does not count (RULE 3)",
146
+ semantic_token=f"requested_outcome:{label}",
147
+ )
124
148
  elif kind == "task_performance":
125
- add("missing_outcome", "medium",
126
- f"frame requests task-performance outcome {label!r} with no "
127
- f"covering finding")
149
+ add(
150
+ "missing_outcome", "medium",
151
+ f"frame requests task-performance outcome {label!r} with no covering finding",
152
+ semantic_token=f"requested_outcome:{label}",
153
+ )
128
154
  elif kind == "learning":
129
- add("missing_outcome", "medium",
130
- f"frame requests learning outcome {label!r} with no covering "
131
- f"learning finding; task performance is not learning (RULE 3)")
155
+ add(
156
+ "missing_outcome", "medium",
157
+ f"frame requests learning outcome {label!r} with no covering learning finding; "
158
+ "task performance is not learning (RULE 3)",
159
+ semantic_token=f"requested_outcome:{label}",
160
+ )
132
161
  else:
133
- add("missing_outcome", "medium",
134
- f"frame requests outcome {label!r} with no covering finding")
135
- claim_outcomes = {c["claim_id"]: c.get("primary_outcome_ids", [])
136
- for c in claims}
162
+ add(
163
+ "missing_outcome", "medium",
164
+ f"frame requests outcome {label!r} with no covering finding",
165
+ semantic_token=f"requested_outcome:{label}",
166
+ )
167
+
168
+ claim_outcomes = {
169
+ c["claim_id"]: c.get("primary_outcome_ids", [])
170
+ for c in claims
171
+ }
137
172
 
138
- # contradiction gaps
139
173
  for syn in syntheses or ():
140
174
  if syn.status == "contested":
141
- add("unresolved_conflict", "high",
175
+ add(
176
+ "unresolved_conflict", "high",
142
177
  f"claim {syn.claim_id} has independent contradictory studies "
143
- f"({', '.join(syn.study_ids)})", [syn.claim_id],
144
- claim_outcomes.get(syn.claim_id, []))
178
+ f"({', '.join(syn.study_ids)})",
179
+ [syn.claim_id],
180
+ claim_outcomes.get(syn.claim_id, []),
181
+ semantic_token=f"claim:{syn.claim_id}",
182
+ )
145
183
 
146
- # methodology weakness / insufficient independence
147
184
  if syntheses:
148
185
  for syn in syntheses:
149
186
  if syn.status == "insufficient" and len(syn.study_ids) < 2:
150
- add("insufficient_sample_independence", "medium",
151
- f"claim {syn.claim_id} rests on fewer than two independent "
152
- f"studies", [syn.claim_id],
153
- claim_outcomes.get(syn.claim_id, []))
154
-
155
- # validate each gap
156
- for g in gaps:
157
- errors = validate_record("knowledge-gap", g)
187
+ add(
188
+ "insufficient_sample_independence", "medium",
189
+ f"claim {syn.claim_id} rests on fewer than two independent studies",
190
+ [syn.claim_id],
191
+ claim_outcomes.get(syn.claim_id, []),
192
+ semantic_token=f"claim:{syn.claim_id}",
193
+ )
194
+
195
+ for gap in gaps:
196
+ errors = validate_record("knowledge-gap", gap)
158
197
  if errors:
159
198
  raise ValueError(f"invalid gap: {errors}")
160
199
  return gaps
@@ -0,0 +1,65 @@
1
+ """Build a transparent, self-describing judge evidence pack from real artifacts."""
2
+ from __future__ import annotations
3
+
4
+ import hashlib
5
+ import json
6
+ import shutil
7
+ from datetime import datetime, timezone
8
+ from pathlib import Path
9
+ from typing import Any
10
+
11
+ from engine.project import ProjectWorkspace
12
+
13
+
14
+ def _sha256(path: Path) -> str:
15
+ h = hashlib.sha256()
16
+ with path.open("rb") as fh:
17
+ for block in iter(lambda: fh.read(65536), b""):
18
+ h.update(block)
19
+ return h.hexdigest()
20
+
21
+
22
+ def export_judge_pack(project: ProjectWorkspace, output_dir: Path) -> dict[str, Any]:
23
+ """Copy available project evidence without inventing unavailable claims.
24
+
25
+ The manifest lists every required judge-pack category and explicitly marks
26
+ missing inputs. This makes the pack suitable for review while keeping its
27
+ limits auditable.
28
+ """
29
+ output_dir = Path(output_dir).resolve()
30
+ output_dir.mkdir(parents=True, exist_ok=True)
31
+ candidates = {
32
+ "project_manifest": project.path / "project.json",
33
+ "graph": project.path / "graph",
34
+ "runs": project.path / "runs",
35
+ "decisions": project.path / "decisions",
36
+ "projections": project.path / "projections",
37
+ "reports": project.path / "reports",
38
+ "pilots": project.path / "pilots",
39
+ }
40
+ copied: list[dict[str, str]] = []
41
+ missing: list[str] = []
42
+ for name, source in candidates.items():
43
+ target = output_dir / name
44
+ if source.is_file():
45
+ shutil.copy2(source, target)
46
+ copied.append({"name": name, "path": target.name, "sha256": _sha256(target)})
47
+ elif source.is_dir() and any(source.rglob("*")):
48
+ shutil.copytree(source, target, dirs_exist_ok=True)
49
+ for item in sorted(path for path in target.rglob("*") if path.is_file()):
50
+ copied.append({"name": name, "path": str(item.relative_to(output_dir)), "sha256": _sha256(item)})
51
+ else:
52
+ missing.append(name)
53
+ manifest = {
54
+ "format": "eduevidence-judge-pack/2026.09",
55
+ "project_id": project.project_id,
56
+ "created_at": datetime.now(timezone.utc).isoformat(),
57
+ "copied_files": copied,
58
+ "missing_categories": missing,
59
+ "limitations": [
60
+ "Only immutable/project-scoped artifacts available at export time are included.",
61
+ "Benchmark, blinded-review and usability evidence must be supplied from completed study artifacts; they are never synthesized by this export.",
62
+ ],
63
+ }
64
+ (output_dir / "manifest.json").write_text(json.dumps(manifest, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
65
+ return manifest
@@ -42,7 +42,9 @@ from functools import lru_cache
42
42
  from pathlib import Path
43
43
  from typing import Any
44
44
 
45
- ROOT = Path(__file__).resolve().parent.parent
45
+ from engine._resources import resource_root
46
+
47
+ ROOT = resource_root()
46
48
  def _resolve_library_path() -> Path:
47
49
  """Repository layout first; wheel-installed share/ layout as fallback."""
48
50
  repo = ROOT / "benchmarks" / "evidence-library.json"
package/engine/living.py CHANGED
@@ -43,7 +43,8 @@ from scripts.validate_schema import SchemaError, validate
43
43
  def _resolve_v4_schema_dir() -> Path:
44
44
  """Repository layout first; wheel-installed share/ layout as fallback
45
45
  (same pattern as engine/contracts._resolve_schema_dir)."""
46
- repo = Path(__file__).resolve().parent.parent / "schemas" / "v4"
46
+ from engine._resources import resource_root
47
+ repo = resource_root() / "schemas" / "v4"
47
48
  if repo.is_dir():
48
49
  return repo
49
50
  import sys
@@ -21,7 +21,9 @@ from engine.ids import new_local_id
21
21
  from engine.library import ResearchLibrary
22
22
  from scripts.validate_schema import SchemaError, validate
23
23
 
24
- _SYNTHESIS_SCHEMA = (Path(__file__).resolve().parent.parent / "schemas" / "v3"
24
+ from engine._resources import resource_root
25
+
26
+ _SYNTHESIS_SCHEMA = (resource_root() / "schemas" / "v3"
25
27
  / "synthesis.schema.json")
26
28
 
27
29