eduevidence 5.2.0 → 6.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (276) hide show
  1. package/README.md +54 -42
  2. package/README.zh-CN.md +51 -28
  3. package/SKILL.md +390 -133
  4. package/agents/openai.yaml +4 -0
  5. package/assets/readme/controlled-execution.svg +34 -0
  6. package/assets/readme/logo.png +0 -0
  7. package/assets/readme/research-workflow.svg +56 -0
  8. package/assets/readme/studio-graph.png +0 -0
  9. package/assets/readme/studio-overview.png +0 -0
  10. package/assets/readme/studio-reports.png +0 -0
  11. package/autoevolve/config.yaml +17 -0
  12. package/autoevolve/program.md +25 -0
  13. package/autoevolve/protected.manifest.yaml +34 -0
  14. package/benchmarks/adversarial/cases.jsonl +7 -0
  15. package/benchmarks/evidence-library.json +5268 -0
  16. package/benchmarks/partitions.json +8 -0
  17. package/docs/architecture.md +220 -0
  18. package/docs/autoresearch-evolution-plan.md +2903 -0
  19. package/docs/autoresearch-implementation-status.md +101 -0
  20. package/docs/demo-storyboard.md +20 -0
  21. package/docs/demo-workplace-ai.md +92 -0
  22. package/docs/demo.md +32 -0
  23. package/docs/install-guide.md +150 -0
  24. package/docs/orchestration-role-model.md +1254 -0
  25. package/docs/release-closeout/README.md +17 -0
  26. package/docs/release-closeout/frontend-acceptance.md +23 -0
  27. package/docs/release-closeout/issues.md +19 -0
  28. package/docs/release-closeout/verification.md +28 -0
  29. package/docs/release-contract.md +108 -0
  30. package/docs/research-studio-guide.zh-CN.md +166 -0
  31. package/eduevidence_cli.py +17 -11
  32. package/engine/_resources.py +13 -0
  33. package/engine/autoevolve/__init__.py +3 -0
  34. package/engine/autoevolve/agent_view.py +167 -0
  35. package/engine/autoevolve/core.py +357 -0
  36. package/engine/autoevolve/events.py +11 -0
  37. package/engine/autoevolve/git_workspace.py +77 -0
  38. package/engine/autoevolve/projection.py +23 -0
  39. package/engine/autoevolve/runner.py +413 -0
  40. package/engine/autoevolve/trust.py +146 -0
  41. package/engine/autoresearch/__init__.py +6 -0
  42. package/engine/autoresearch/commit.py +132 -0
  43. package/engine/autoresearch/contracts.py +126 -0
  44. package/engine/autoresearch/controller.py +207 -0
  45. package/engine/autoresearch/events.py +12 -0
  46. package/engine/autoresearch/gap_priority.py +168 -0
  47. package/engine/autoresearch/projection.py +30 -0
  48. package/engine/autoresearch/research_memory.py +59 -0
  49. package/engine/autoresearch/saturation.py +91 -0
  50. package/engine/briefs.py +2 -1
  51. package/engine/capabilities.py +1 -0
  52. package/engine/contracts.py +3 -1
  53. package/engine/evidencecore.py +7 -5
  54. package/engine/gaps.py +90 -51
  55. package/engine/judge_pack.py +65 -0
  56. package/engine/library_builtin.py +3 -1
  57. package/engine/living.py +2 -1
  58. package/engine/meta_synthesis.py +3 -1
  59. package/engine/orchestration.py +460 -0
  60. package/engine/pilot.py +2 -1
  61. package/engine/project.py +2 -2
  62. package/engine/research_service.py +113 -0
  63. package/engine/studio_read_model.py +400 -0
  64. package/engine/tribunal.py +1 -2
  65. package/engine/update.py +1 -0
  66. package/engine/versions.py +1 -1
  67. package/engine/worker_result.py +109 -0
  68. package/engine/workflows.py +70 -0
  69. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1720 -0
  70. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
  71. package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
  72. package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
  73. package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
  74. package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
  75. package/examples/ai-coding-assistant-evidence/frame.json +48 -0
  76. package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
  77. package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
  78. package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
  79. package/examples/ai-coding-assistant-evidence/report_spec.json +219 -0
  80. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2614 -0
  81. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2614 -0
  82. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2614 -0
  83. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2614 -0
  84. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2614 -0
  85. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1720 -0
  86. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1720 -0
  87. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1720 -0
  88. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1720 -0
  89. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1720 -0
  90. package/examples/ai-coding-assistant-evidence/result.json +1453 -0
  91. package/examples/ai-coding-assistant-evidence/result.zh.json +1453 -0
  92. package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
  93. package/examples/ai-coding-assistant-evidence/verdict.json +103 -0
  94. package/examples/workplace-ai-assistant/claims.jsonl +4 -0
  95. package/examples/workplace-ai-assistant/evaluation.json +19 -0
  96. package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
  97. package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
  98. package/examples/workplace-ai-assistant/frame.json +41 -0
  99. package/examples/workplace-ai-assistant/intervention.json +27 -0
  100. package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
  101. package/examples/workplace-ai-assistant/methodology.json +60 -0
  102. package/examples/workplace-ai-assistant/report_spec.json +55 -0
  103. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2484 -0
  104. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2484 -0
  105. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2484 -0
  106. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2484 -0
  107. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2484 -0
  108. package/examples/workplace-ai-assistant/result.json +553 -0
  109. package/examples/workplace-ai-assistant/result.zh.json +553 -0
  110. package/examples/workplace-ai-assistant/search_log.json +19 -0
  111. package/examples/workplace-ai-assistant/sources.jsonl +3 -0
  112. package/examples/workplace-ai-assistant/validation_result.json +9 -0
  113. package/examples/workplace-ai-assistant/verdict.json +52 -0
  114. package/install.sh +7 -7
  115. package/integrations/orchestration_dispatch.py +146 -0
  116. package/package.json +37 -3
  117. package/pyproject.toml +11 -20
  118. package/references/autoresearch.md +30 -0
  119. package/references/evaluation-policy.md +24 -0
  120. package/references/orchestration.md +22 -0
  121. package/references/scientific-invariants.md +19 -0
  122. package/retrieval/audit.py +154 -0
  123. package/schemas/intervention.schema.json +106 -21
  124. package/schemas/report-result.schema.json +9 -1
  125. package/schemas/v2/project.schema.json +2 -2
  126. package/schemas/v2/run.schema.json +1 -1
  127. package/schemas/vNext/autoevolve-session.schema.json +1 -0
  128. package/schemas/vNext/eval-snapshot.schema.json +1 -0
  129. package/schemas/vNext/execution-plan.schema.json +1 -0
  130. package/schemas/vNext/gap-priority.schema.json +1 -0
  131. package/schemas/vNext/negative-search-record.schema.json +1 -0
  132. package/schemas/vNext/research-iteration.schema.json +1 -0
  133. package/schemas/vNext/research-strategy.schema.json +1 -0
  134. package/schemas/vNext/skill-experiment.schema.json +1 -0
  135. package/schemas/vNext/task-spec.schema.json +1 -0
  136. package/schemas/vNext/worker-result.schema.json +1 -0
  137. package/scripts/benchmark_judge.py +2 -2
  138. package/scripts/benchmark_v3.py +26 -43
  139. package/scripts/build_esl_artifacts.py +2 -2
  140. package/scripts/build_evidence_library.py +2 -2
  141. package/scripts/build_gh_pages.py +98 -0
  142. package/scripts/build_readme_diagrams.py +72 -0
  143. package/scripts/build_report_variants.py +85 -0
  144. package/scripts/check_autoresearch_invariants.py +95 -0
  145. package/scripts/daily_evolve.py +30 -0
  146. package/scripts/dashboard_server.py +130 -101
  147. package/scripts/did_regression.py +5 -30
  148. package/scripts/enrich_projects_human_and_lieflat.py +1 -1
  149. package/scripts/generate_metrics.py +4 -3
  150. package/scripts/generate_new_projects.py +1 -1
  151. package/scripts/orchestrator.py +172 -18
  152. package/scripts/rebake_all_5themes.py +1 -2
  153. package/scripts/research_auto_cli.py +475 -0
  154. package/scripts/run_workspace.py +17 -7
  155. package/scripts/search_provenance.py +64 -0
  156. package/scripts/serve_web.py +9 -10
  157. package/scripts/skill_lint.py +1 -1
  158. package/scripts/skill_payload.py +78 -0
  159. package/scripts/validate_schema.py +15 -1
  160. package/scripts/vnext_cli.py +133 -0
  161. package/setup.py +12 -0
  162. package/skill/roles/registry.yaml +45 -0
  163. package/skill/sub-skills/report-generation/SKILL.md +12 -6
  164. package/skill/task-briefs/applicability.md +3 -0
  165. package/skill/task-briefs/projection.md +3 -0
  166. package/skill/workflows/decision-and-pilot.md +10 -0
  167. package/skill/workflows/evaluate-and-update.md +10 -0
  168. package/skill/workflows/evidence-review.md +13 -0
  169. package/visualization/eduevidence-report/assets/base.css +2 -2
  170. package/visualization/eduevidence-report/assets/reader.css +752 -0
  171. package/visualization/eduevidence-report/assets/reader.js +132 -0
  172. package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
  173. package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
  174. package/visualization/eduevidence-report/scripts/build_report.py +58 -65
  175. package/visualization/eduevidence-report/scripts/lieflat_engine.py +24 -100
  176. package/visualization/eduevidence-report/themes/academic.css +1 -1
  177. package/visualization/eduevidence-report/themes/claude.css +1 -1
  178. package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
  179. package/visualization/eduevidence-report/themes/datalab.css +2 -2
  180. package/visualization/eduevidence-report/themes/presentation.css +2 -2
  181. package/web/README.md +18 -0
  182. package/web/index.html +53 -0
  183. package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
  184. package/web/studio/assets/index-CzXocaGv.css +1 -0
  185. package/web/studio/assets/index-pa7jD7n4.js +230 -0
  186. package/web/studio/config.json +1 -0
  187. package/web/studio/index.html +14 -0
  188. package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
  189. package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
  190. package/engine/__pycache__/bias.cpython-312.pyc +0 -0
  191. package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
  192. package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
  193. package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
  194. package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
  195. package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
  196. package/engine/__pycache__/events.cpython-312.pyc +0 -0
  197. package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
  198. package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
  199. package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
  200. package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
  201. package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
  202. package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
  203. package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
  204. package/engine/__pycache__/ids.cpython-312.pyc +0 -0
  205. package/engine/__pycache__/library.cpython-312.pyc +0 -0
  206. package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
  207. package/engine/__pycache__/living.cpython-312.pyc +0 -0
  208. package/engine/__pycache__/log.cpython-312.pyc +0 -0
  209. package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
  210. package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
  211. package/engine/__pycache__/migration.cpython-312.pyc +0 -0
  212. package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
  213. package/engine/__pycache__/paths.cpython-312.pyc +0 -0
  214. package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
  215. package/engine/__pycache__/planner.cpython-312.pyc +0 -0
  216. package/engine/__pycache__/project.cpython-312.pyc +0 -0
  217. package/engine/__pycache__/projections.cpython-312.pyc +0 -0
  218. package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
  219. package/engine/__pycache__/run.cpython-312.pyc +0 -0
  220. package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
  221. package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
  222. package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
  223. package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
  224. package/engine/__pycache__/update.cpython-312.pyc +0 -0
  225. package/engine/__pycache__/versions.cpython-312.pyc +0 -0
  226. package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
  227. package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
  228. package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
  229. package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
  230. package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
  231. package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
  232. package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
  233. package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
  234. package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
  235. package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
  236. package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
  237. package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
  238. package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
  239. package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
  240. package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
  241. package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
  242. package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
  243. package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
  244. package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
  245. package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
  246. package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
  247. package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
  248. package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
  249. package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
  250. package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
  251. package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
  252. package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
  253. package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
  254. package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
  255. package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
  256. package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
  257. package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
  258. package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
  259. package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
  260. package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
  261. package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
  262. package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
  263. package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
  264. package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
  265. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
  266. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
  267. package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
  268. package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
  269. package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
  270. package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
  271. package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
  272. package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
  273. package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
  274. package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
  275. package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
  276. package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
@@ -0,0 +1,357 @@
1
+ from __future__ import annotations
2
+
3
+ import fnmatch
4
+ import hashlib
5
+ import json
6
+ from dataclasses import asdict, dataclass, field
7
+ from pathlib import Path
8
+
9
+ PROTECTED_DEFAULTS = (
10
+ "autoevolve/protected.manifest.yaml",
11
+ "benchmarks/annotations/**",
12
+ "benchmarks/holdout/**",
13
+ "benchmarks/evaluator/**",
14
+ "schemas/**",
15
+ "references/scientific-invariants.md",
16
+ "scripts/pre_verdict_gate.py",
17
+ "scripts/compute_confidence.py",
18
+ "scripts/check_autoresearch_invariants.py",
19
+ "engine/graph_store.py",
20
+ "engine/study_design.py",
21
+ "engine/autoevolve/**",
22
+ ".github/workflows/autoresearch-gates.yml",
23
+ )
24
+ SAFE_DEFAULTS = (
25
+ "skill/workflows/**",
26
+ "skill/agents/**",
27
+ "skill/sub-skills/**",
28
+ "retrieval/**",
29
+ "references/presentation/**",
30
+ )
31
+ CONTROLLED_DEFAULTS = (
32
+ "engine/semantics.py",
33
+ "engine/gaps.py",
34
+ "scripts/complexity_gate.py",
35
+ "engine/orchestration.py",
36
+ )
37
+
38
+
39
+ @dataclass(frozen=True)
40
+ class EvalSnapshot:
41
+ eval_id: str
42
+ hard_gates_passed: bool
43
+ science_score: float
44
+ research_score: float
45
+ robustness: float
46
+ cost: float
47
+ latency: float
48
+ complexity: float
49
+ repeats: int = 1
50
+ noise_floor: float = 0.0
51
+ dev_passed: bool = False
52
+ holdout_passed: bool = False
53
+ adversarial_passed: bool = False
54
+ holdout_isolation_verified: bool = False
55
+ eval_suite_hash: str = ""
56
+
57
+
58
+ @dataclass
59
+ class SkillExperiment:
60
+ experiment_id: str
61
+ session_id: str
62
+ parent_skill_revision: str
63
+ hypothesis: str
64
+ mutation_scope: tuple[str, ...]
65
+ changed_files: list[str] = field(default_factory=list)
66
+ candidate_commit: str | None = None
67
+ baseline_eval_id: str | None = None
68
+ candidate_eval_id: str | None = None
69
+ protected_hash_before: str | None = None
70
+ protected_hash_after: str | None = None
71
+ status: str = "created"
72
+ promotion_reason: str = ""
73
+ complexity_delta: float = 0.0
74
+
75
+
76
+ def _matches(path: str, patterns: tuple[str, ...]) -> bool:
77
+ normalized = path.replace("\\", "/")
78
+ return any(
79
+ fnmatch.fnmatch(normalized, pattern)
80
+ or (pattern.endswith("/**") and normalized.startswith(pattern[:-3]))
81
+ for pattern in patterns
82
+ )
83
+
84
+
85
+ def _parse_manifest(path: Path) -> tuple[tuple[str, ...], tuple[str, ...], tuple[str, ...]]:
86
+ """Parse the intentionally tiny manifest subset without a YAML dependency."""
87
+ protected: list[str] = []
88
+ safe: list[str] = []
89
+ controlled: list[str] = []
90
+ section = ""
91
+ subsection = ""
92
+ for raw in path.read_text(encoding="utf-8").splitlines():
93
+ line = raw.split("#", 1)[0].rstrip()
94
+ if not line.strip():
95
+ continue
96
+ stripped = line.strip()
97
+ indent = len(line) - len(line.lstrip(" "))
98
+ if indent == 0 and stripped.endswith(":"):
99
+ section = stripped[:-1]
100
+ subsection = ""
101
+ continue
102
+ if section == "mutable" and indent == 2 and stripped.endswith(":"):
103
+ subsection = stripped[:-1]
104
+ continue
105
+ if stripped.startswith("- "):
106
+ value = stripped[2:].strip().strip('"\'')
107
+ if section == "protected":
108
+ protected.append(value)
109
+ elif section == "mutable" and subsection == "safe":
110
+ safe.append(value)
111
+ elif section == "mutable" and subsection == "controlled":
112
+ controlled.append(value)
113
+ if not protected or not safe:
114
+ raise ValueError(f"invalid autoresearch manifest: {path}")
115
+ return tuple(protected), tuple(safe), tuple(controlled)
116
+
117
+
118
+ class ProtectedManifest:
119
+ def __init__(
120
+ self,
121
+ patterns=PROTECTED_DEFAULTS,
122
+ *,
123
+ safe_patterns=SAFE_DEFAULTS,
124
+ controlled_patterns=CONTROLLED_DEFAULTS,
125
+ ):
126
+ self.patterns = tuple(patterns)
127
+ self.safe_patterns = tuple(safe_patterns)
128
+ self.controlled_patterns = tuple(controlled_patterns)
129
+
130
+ @classmethod
131
+ def from_repo(cls, root: str | Path) -> "ProtectedManifest":
132
+ path = Path(root) / "autoevolve" / "protected.manifest.yaml"
133
+ if not path.is_file():
134
+ return cls()
135
+ protected, safe, controlled = _parse_manifest(path)
136
+ merged_protected = tuple(dict.fromkeys((*protected, *PROTECTED_DEFAULTS)))
137
+ return cls(
138
+ merged_protected,
139
+ safe_patterns=safe,
140
+ controlled_patterns=controlled,
141
+ )
142
+
143
+ def is_protected(self, path: str) -> bool:
144
+ return _matches(path, self.patterns)
145
+
146
+ def classify(self, path: str) -> str:
147
+ if self.is_protected(path):
148
+ return "protected"
149
+ if _matches(path, self.safe_patterns):
150
+ return "safe"
151
+ if _matches(path, self.controlled_patterns):
152
+ return "controlled"
153
+ return "unknown"
154
+
155
+ def validate_changes(self, changed: list[str]) -> tuple[bool, list[str]]:
156
+ bad = [path for path in changed if self.is_protected(path)]
157
+ return (not bad, bad)
158
+
159
+ def validate_mutation_scope(
160
+ self,
161
+ changed: list[str],
162
+ *,
163
+ mutation_tiers: tuple[str, ...] | list[str],
164
+ allow_controlled: bool,
165
+ ) -> tuple[bool, list[str]]:
166
+ tiers = set(mutation_tiers)
167
+ bad: list[str] = []
168
+ for path in changed:
169
+ kind = self.classify(path)
170
+ allowed = kind == "safe" and "safe" in tiers
171
+ allowed = allowed or (
172
+ kind == "controlled" and allow_controlled and "controlled" in tiers
173
+ )
174
+ if not allowed:
175
+ bad.append(path)
176
+ return (not bad, bad)
177
+
178
+ def hash_tree(self, root: str | Path) -> str:
179
+ root = Path(root)
180
+ digest = hashlib.sha256()
181
+ files = []
182
+ for file in root.rglob("*"):
183
+ if file.is_file() and self.is_protected(file.relative_to(root).as_posix()):
184
+ files.append(file)
185
+ for file in sorted(files):
186
+ rel = file.relative_to(root).as_posix()
187
+ digest.update(rel.encode())
188
+ digest.update(b"\0")
189
+ digest.update(file.read_bytes())
190
+ digest.update(b"\0")
191
+ return digest.hexdigest()
192
+
193
+
194
+ def _promotion_evidence_ready(baseline: EvalSnapshot, candidate: EvalSnapshot) -> tuple[bool, str]:
195
+ if not baseline.eval_suite_hash or baseline.eval_suite_hash != candidate.eval_suite_hash:
196
+ return False, "baseline/candidate eval suite hash missing or mismatched"
197
+ required = (
198
+ baseline.dev_passed,
199
+ baseline.holdout_passed,
200
+ baseline.adversarial_passed,
201
+ baseline.holdout_isolation_verified,
202
+ candidate.dev_passed,
203
+ candidate.holdout_passed,
204
+ candidate.adversarial_passed,
205
+ candidate.holdout_isolation_verified,
206
+ )
207
+ if not all(required):
208
+ return False, "DEV/HOLDOUT/adversarial gates and holdout isolation are required for automatic KEEP"
209
+ return True, ""
210
+
211
+
212
+ def promote(
213
+ baseline: EvalSnapshot,
214
+ candidate: EvalSnapshot,
215
+ *,
216
+ simplicity_tolerance: float = 0.0,
217
+ efficiency_tolerance: float = 0.05,
218
+ minimum_repeats: int = 3,
219
+ ) -> tuple[str, str]:
220
+ """Constraint-first promotion; automatic KEEP is deliberately conservative."""
221
+ if not candidate.hard_gates_passed:
222
+ return "REJECT", "L0 hard gate failed"
223
+ if candidate.science_score < baseline.science_score:
224
+ return "REJECT", "scientific correctness regressed"
225
+ if min(baseline.repeats, candidate.repeats) < minimum_repeats:
226
+ return "RETEST", f"automatic promotion requires >= {minimum_repeats} repeated runs"
227
+
228
+ delta = candidate.research_score - baseline.research_score
229
+ noise = max(baseline.noise_floor, candidate.noise_floor)
230
+ if delta < -noise:
231
+ return "REJECT", "research-quality regression exceeds empirical noise floor"
232
+
233
+ regressions = []
234
+ if candidate.robustness < baseline.robustness:
235
+ regressions.append("robustness")
236
+ if baseline.cost > 0 and candidate.cost > baseline.cost * (1 + efficiency_tolerance):
237
+ regressions.append("cost")
238
+ if baseline.latency > 0 and candidate.latency > baseline.latency * (1 + efficiency_tolerance):
239
+ regressions.append("latency")
240
+ if candidate.complexity > baseline.complexity + simplicity_tolerance:
241
+ regressions.append("complexity")
242
+
243
+ if abs(delta) <= noise:
244
+ if regressions:
245
+ return "REJECT", "within noise floor with regression: " + ",".join(regressions)
246
+ if candidate.complexity < baseline.complexity - simplicity_tolerance:
247
+ ready, why = _promotion_evidence_ready(baseline, candidate)
248
+ if not ready:
249
+ return "HUMAN_REVIEW", why
250
+ return "KEEP", "equivalent research quality with simpler implementation"
251
+ return "RETEST", "candidate delta is within empirical noise floor"
252
+
253
+ if delta > noise and not regressions:
254
+ ready, why = _promotion_evidence_ready(baseline, candidate)
255
+ if not ready:
256
+ return "HUMAN_REVIEW", why
257
+ return "KEEP", "material research-quality improvement without Pareto regression"
258
+ return "HUMAN_REVIEW", "Pareto trade-off: " + ",".join(regressions or ["mixed metrics"])
259
+
260
+
261
+ class ExperimentLog:
262
+ HEADER = (
263
+ "experiment_id",
264
+ "parent_revision",
265
+ "candidate_commit",
266
+ "scope",
267
+ "hypothesis",
268
+ "eval_suite_hash",
269
+ "repeats",
270
+ "dev_passed",
271
+ "holdout_passed",
272
+ "adversarial_passed",
273
+ "holdout_isolation_verified",
274
+ "hard_gates",
275
+ "science_score",
276
+ "research_score",
277
+ "robustness",
278
+ "cost",
279
+ "latency",
280
+ "complexity_delta",
281
+ "status",
282
+ "description",
283
+ )
284
+
285
+ def __init__(self, root: str | Path):
286
+ self.root = Path(root)
287
+ self.root.mkdir(parents=True, exist_ok=True)
288
+ self.tsv = self.root / "results.tsv"
289
+ self.jsonl = self.root / "experiments.jsonl"
290
+ if not self.tsv.exists():
291
+ self.tsv.write_text("\t".join(self.HEADER) + "\n", encoding="utf-8")
292
+
293
+ def append(
294
+ self,
295
+ experiment: SkillExperiment,
296
+ *,
297
+ candidate: EvalSnapshot | None = None,
298
+ description: str = "",
299
+ ) -> None:
300
+ row = [
301
+ experiment.experiment_id,
302
+ experiment.parent_skill_revision,
303
+ experiment.candidate_commit or "",
304
+ ",".join(experiment.mutation_scope),
305
+ experiment.hypothesis,
306
+ str(candidate.eval_suite_hash if candidate else ""),
307
+ str(candidate.repeats if candidate else ""),
308
+ str(candidate.dev_passed if candidate else ""),
309
+ str(candidate.holdout_passed if candidate else ""),
310
+ str(candidate.adversarial_passed if candidate else ""),
311
+ str(candidate.holdout_isolation_verified if candidate else ""),
312
+ str(candidate.hard_gates_passed if candidate else ""),
313
+ str(candidate.science_score if candidate else ""),
314
+ str(candidate.research_score if candidate else ""),
315
+ str(candidate.robustness if candidate else ""),
316
+ str(candidate.cost if candidate else ""),
317
+ str(candidate.latency if candidate else ""),
318
+ str(experiment.complexity_delta),
319
+ experiment.status,
320
+ description,
321
+ ]
322
+ with self.tsv.open("a", encoding="utf-8") as handle:
323
+ handle.write("\t".join(value.replace("\t", " ") for value in row) + "\n")
324
+ with self.jsonl.open("a", encoding="utf-8") as handle:
325
+ handle.write(json.dumps(asdict(experiment), ensure_ascii=False, sort_keys=True) + "\n")
326
+
327
+
328
+ class PlateauTracker:
329
+ def __init__(self, limit: int = 5):
330
+ self.limit = limit
331
+
332
+ def plateau(self, statuses: list[str]) -> bool:
333
+ valid = [status for status in statuses if status not in {"CRASH", "INVALID", "RETEST"}]
334
+ return len(valid) >= self.limit and all(status != "KEEP" for status in valid[-self.limit :])
335
+
336
+
337
+ @dataclass(frozen=True)
338
+ class DailyProfile:
339
+ max_experiments: int = 20
340
+ max_cost_usd: float = 5.0
341
+ max_wall_minutes: int = 180
342
+ mutation_tiers: tuple[str, ...] = ("safe",)
343
+ allow_controlled: bool = False
344
+ promotion: str = "branch_only"
345
+
346
+ def validate(self) -> None:
347
+ if not 1 <= self.max_experiments <= 50:
348
+ raise ValueError("daily max_experiments must be 1..50")
349
+ if self.max_cost_usd <= 0 or self.max_wall_minutes <= 0:
350
+ raise ValueError("daily budget must be positive")
351
+ if self.promotion != "branch_only":
352
+ raise ValueError("daily mode is branch_only")
353
+ unknown = set(self.mutation_tiers) - {"safe", "controlled"}
354
+ if unknown:
355
+ raise ValueError(f"unknown mutation tiers: {sorted(unknown)}")
356
+ if "controlled" in self.mutation_tiers and not self.allow_controlled:
357
+ raise ValueError("controlled mutation tier requires allow_controlled=true")
@@ -0,0 +1,11 @@
1
+ SKILL_AUTORESEARCH_EVENTS = (
2
+ "autoevolve.session.started",
3
+ "autoevolve.experiment.created",
4
+ "autoevolve.candidate.built",
5
+ "autoevolve.eval.completed",
6
+ "autoevolve.candidate.kept",
7
+ "autoevolve.candidate.rejected",
8
+ "autoevolve.candidate.retest",
9
+ "autoevolve.plateau",
10
+ "autoevolve.session.completed",
11
+ )
@@ -0,0 +1,77 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ import subprocess
5
+ from dataclasses import dataclass
6
+ from pathlib import Path
7
+
8
+
9
+ def _run(repo: Path, *args: str, capture: bool = False) -> str:
10
+ completed = subprocess.run(
11
+ ["git", *args], cwd=repo, check=True, text=True, capture_output=capture
12
+ )
13
+ return completed.stdout.rstrip("\n") if capture else ""
14
+
15
+
16
+ def safe_tag(tag: str) -> str:
17
+ clean = re.sub(r"[^A-Za-z0-9._-]+", "-", tag).strip("-")
18
+ if not clean or clean in {".", ".."}:
19
+ raise ValueError("invalid run tag")
20
+ return clean[:80]
21
+
22
+
23
+ @dataclass(frozen=True)
24
+ class GitWorkspace:
25
+ repo: Path
26
+ path: Path
27
+ branch: str
28
+
29
+ @classmethod
30
+ def create(cls, repo: str | Path, tag: str):
31
+ repo = Path(repo).resolve()
32
+ tag = safe_tag(tag)
33
+ branch = f"autoresearch/{tag}"
34
+ current = _run(repo, "branch", "--show-current", capture=True)
35
+ if current.startswith("autoresearch/"):
36
+ raise ValueError("create the session from a non-autoresearch base branch")
37
+ path = repo / ".autoevolve-worktrees" / tag
38
+ path.parent.mkdir(parents=True, exist_ok=True)
39
+ if path.exists():
40
+ raise FileExistsError(path)
41
+ _run(repo, "worktree", "add", "-b", branch, str(path), "HEAD")
42
+ return cls(repo, path, branch)
43
+
44
+ def head(self) -> str:
45
+ return _run(self.path, "rev-parse", "HEAD", capture=True)
46
+
47
+ def diff(self) -> str:
48
+ return _run(self.path, "diff", "--binary", "HEAD", capture=True)
49
+
50
+ def changed_files(self) -> list[str]:
51
+ output = _run(self.path, "status", "--porcelain", capture=True)
52
+ return [line[3:] for line in output.splitlines() if line.strip()]
53
+
54
+ def restore(self) -> None:
55
+ output = _run(self.path, "status", "--porcelain", capture=True)
56
+ untracked = [line[3:] for line in output.splitlines() if line.startswith("?? ")]
57
+ _run(self.path, "restore", "--staged", "--worktree", ".")
58
+ for relative in untracked:
59
+ target = (self.path / relative).resolve()
60
+ if self.path not in target.parents and target != self.path:
61
+ raise ValueError("unsafe untracked path")
62
+ if target.is_file() or target.is_symlink():
63
+ target.unlink(missing_ok=True)
64
+ elif target.is_dir():
65
+ import shutil
66
+ shutil.rmtree(target)
67
+
68
+ def commit(self, message: str) -> str:
69
+ _run(self.path, "add", "-A")
70
+ _run(self.path, "commit", "-m", message)
71
+ return _run(self.path, "rev-parse", "HEAD", capture=True)
72
+
73
+ def push(self, remote: str = "origin") -> None:
74
+ """Push only this experiment branch. Never force-push or merge."""
75
+ if not self.branch.startswith("autoresearch/"):
76
+ raise ValueError("only autoresearch branches may be pushed by autoevolve")
77
+ _run(self.path, "push", "-u", remote, self.branch)
@@ -0,0 +1,23 @@
1
+ from __future__ import annotations
2
+ from typing import Any
3
+ from .core import PlateauTracker
4
+
5
+
6
+ def skill_evolution_projection(
7
+ *,
8
+ baseline: dict[str, Any] | None,
9
+ best: dict[str, Any] | None,
10
+ experiments: list[dict[str, Any]],
11
+ protected_integrity: bool = True,
12
+ ) -> dict[str, Any]:
13
+ """Projection-only developer state; never mutates repository or user projects."""
14
+ statuses = [str(item.get("status", "")) for item in experiments]
15
+ return {
16
+ "baseline": baseline,
17
+ "best": best,
18
+ "experiments": experiments,
19
+ "experiment_count": len(experiments),
20
+ "plateau": PlateauTracker().plateau(statuses),
21
+ "protected_integrity": protected_integrity,
22
+ "promotion": "branch_only",
23
+ }