world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,509 @@
1
+ """Sequential population optimization of complete harness source trees.
2
+
3
+ `optimize` is the paper's outer loop: score the fixed seed once, then consume a fixed number of
4
+ sequential proposal slots. Each slot asks the proposer for one complete candidate source tree.
5
+ An invalid proposal (`CandidateProposalError`) consumes its slot as recorded evidence and the
6
+ loop continues; scorer and infrastructure exceptions PROPAGATE, because a missing evaluation can
7
+ never be reinterpreted as a reward. Selection is the maximum mean per-task score with the
8
+ earliest candidate winning ties.
9
+
10
+ Durable state is the run directory. After every boundary (the scored seed, one scored proposal,
11
+ or one consumed invalid slot) the outcome's evidence lands under `candidates/candidate-NNNN/`
12
+ (`source/`, `report.json`, and `proposal.json` or `error.json`) and the ordered index is
13
+ committed by an atomic tmp+rename of `state.json`. An ACCEPTED proposal is additionally
14
+ checkpointed (`source/` + `proposal.json` + `pending.json`) BEFORE its evaluation starts, so a
15
+ crash mid-score resumes by rescoring the exact same candidate instead of paying a fresh proposer
16
+ turn whose different doc hash would also orphan the part-paid evaluator job. Resuming reloads
17
+ state and continues at the first missing slot, so an interrupted run re-pays at most one
18
+ boundary (and the harbor scorer's own trial-level resume usually far less).
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import json
24
+ import logging
25
+ import os
26
+ import shutil
27
+ from collections.abc import Callable, Sequence
28
+ from dataclasses import dataclass
29
+ from pathlib import Path
30
+ from typing import Protocol
31
+
32
+ from pydantic import JsonValue
33
+
34
+ from wmo.harness.doc import HarnessDoc
35
+ from wmo.harness.runtime import HarnessSearchCancelled
36
+ from wmo.harness.scoring import Scorer, ScoreReport
37
+ from wmo.harness.source_tree import HarnessSourceFile, HarnessSourceTree
38
+
39
+ logger = logging.getLogger(__name__)
40
+
41
+ _STATE_FILE = "state.json"
42
+ _CANDIDATES_DIR = "candidates"
43
+ _PENDING_FILE = "pending.json"
44
+
45
+
46
+ def candidate_slot_id(slot: int) -> str:
47
+ """The fixed candidate identity for one slot (`candidate-0000` is the seed)."""
48
+ if isinstance(slot, bool) or not isinstance(slot, int) or slot < 0:
49
+ raise ValueError("slot must be a nonnegative integer")
50
+ return f"candidate-{slot:04d}"
51
+
52
+
53
+ @dataclass(frozen=True)
54
+ class EvaluatedCandidate:
55
+ """One complete source candidate paired with its immutable score report."""
56
+
57
+ candidate_id: str
58
+ source: HarnessSourceTree
59
+ report: ScoreReport
60
+
61
+ def __post_init__(self) -> None:
62
+ if self.source.to_doc(self.candidate_id).doc_hash != self.report.doc_hash:
63
+ raise ValueError(
64
+ f"score report for {self.candidate_id!r} does not match its source tree"
65
+ )
66
+
67
+ @property
68
+ def candidate(self) -> HarnessDoc:
69
+ """Reparse the complete source into its validated harness document."""
70
+ return self.source.to_doc(self.candidate_id)
71
+
72
+ @property
73
+ def score(self) -> float:
74
+ """The selection objective: mean of per-task pass rates."""
75
+ return self.report.score
76
+
77
+
78
+ @dataclass(frozen=True)
79
+ class CandidateProposal:
80
+ """One complete, host-captured and reparsed candidate proposal."""
81
+
82
+ candidate_id: str
83
+ source: HarnessSourceTree
84
+ candidate: HarnessDoc
85
+
86
+ def __post_init__(self) -> None:
87
+ if self.candidate.name != self.candidate_id:
88
+ raise ValueError("proposal document name does not match its candidate_id")
89
+ if self.source.to_doc(self.candidate_id).doc_hash != self.candidate.doc_hash:
90
+ raise ValueError("proposal source does not match its candidate document")
91
+
92
+
93
+ class CandidateProposalError(RuntimeError):
94
+ """One proposal turn that did not publish a valid complete candidate.
95
+
96
+ This is a CANDIDATE outcome: the loop records it and the slot is consumed. Raw evidence
97
+ (the request, events, raw snapshot, and error) lives under ``evidence_dir`` when set.
98
+ """
99
+
100
+ def __init__(self, candidate_id: str, reason: str, *, evidence_dir: str = "") -> None:
101
+ super().__init__(f"{candidate_id}: {reason}")
102
+ self.candidate_id = candidate_id
103
+ self.reason = reason
104
+ self.evidence_dir = evidence_dir
105
+
106
+
107
+ class CandidateProposer(Protocol):
108
+ """Produce exactly one complete candidate for one slot from the evaluated population."""
109
+
110
+ def propose(
111
+ self,
112
+ population: Sequence[EvaluatedCandidate],
113
+ *,
114
+ slot: int,
115
+ should_cancel: Callable[[], bool] | None = None,
116
+ ) -> CandidateProposal: ...
117
+
118
+
119
+ @dataclass(frozen=True)
120
+ class SlotOutcome:
121
+ """One consumed slot: either a scored candidate or a recorded invalid proposal."""
122
+
123
+ slot: int
124
+ candidate_id: str
125
+ evaluated: EvaluatedCandidate | None = None
126
+ reason: str = ""
127
+ evidence_dir: str = ""
128
+
129
+ def __post_init__(self) -> None:
130
+ if self.candidate_id != candidate_slot_id(self.slot):
131
+ raise ValueError("slot outcome candidate_id does not match its slot")
132
+ if (self.evaluated is None) == (not self.reason):
133
+ raise ValueError("a slot outcome is either evaluated or carries an invalid reason")
134
+
135
+
136
+ @dataclass(frozen=True)
137
+ class PopulationResult:
138
+ """Every consumed slot, the evaluated population, and the current score winner."""
139
+
140
+ outcomes: tuple[SlotOutcome, ...]
141
+ population: tuple[EvaluatedCandidate, ...]
142
+ best: EvaluatedCandidate
143
+ completed: bool
144
+
145
+ @property
146
+ def best_score(self) -> float:
147
+ return self.best.score
148
+
149
+
150
+ def write_json_atomic(path: Path, value: JsonValue) -> None:
151
+ """Write deterministic JSON through a same-directory fsynced tmp+rename.
152
+
153
+ The temp file is fsynced before the rename and the directory after it: without both, a
154
+ power loss can persist the rename while the data blocks are lost, leaving a truncated or
155
+ empty state file behind an apparently successful commit.
156
+ """
157
+ temporary = path.with_name(f"{path.name}.tmp")
158
+ temporary.parent.mkdir(parents=True, exist_ok=True)
159
+ payload = json.dumps(value, indent=2, sort_keys=True, ensure_ascii=False) + "\n"
160
+ with temporary.open("w", encoding="utf-8") as handle:
161
+ handle.write(payload)
162
+ handle.flush()
163
+ os.fsync(handle.fileno())
164
+ temporary.replace(path)
165
+ directory = os.open(path.parent, os.O_RDONLY)
166
+ try:
167
+ os.fsync(directory)
168
+ finally:
169
+ os.close(directory)
170
+
171
+
172
+ class PopulationRunState:
173
+ """Durable per-boundary population state under one run directory."""
174
+
175
+ def __init__(self, run_dir: str | Path) -> None:
176
+ self.run_dir = Path(run_dir)
177
+
178
+ def candidate_dir(self, candidate_id: str) -> Path:
179
+ return self.run_dir / _CANDIDATES_DIR / candidate_id
180
+
181
+ def load(self) -> tuple[SlotOutcome, ...]:
182
+ """Reload every committed slot outcome, re-verifying candidate evidence integrity."""
183
+ path = self.run_dir / _STATE_FILE
184
+ if not path.exists():
185
+ return ()
186
+ raw = json.loads(path.read_text(encoding="utf-8"))
187
+ entries = raw.get("outcomes")
188
+ if not isinstance(entries, list):
189
+ raise ValueError(f"{path} does not contain an outcomes list")
190
+ outcomes: list[SlotOutcome] = []
191
+ for index, entry in enumerate(entries):
192
+ candidate_id = candidate_slot_id(index)
193
+ if not isinstance(entry, dict):
194
+ raise ValueError(f"{path} outcome at slot {index} is not an object")
195
+ if entry.get("slot") != index or entry.get("candidate_id") != candidate_id:
196
+ raise ValueError(f"{path} outcomes are not contiguous at slot {index}")
197
+ if entry.get("kind") == "invalid":
198
+ outcomes.append(
199
+ SlotOutcome(
200
+ slot=index,
201
+ candidate_id=candidate_id,
202
+ reason=str(entry.get("reason") or "invalid proposal"),
203
+ evidence_dir=str(entry.get("evidence_dir") or ""),
204
+ )
205
+ )
206
+ continue
207
+ directory = self.candidate_dir(candidate_id)
208
+ report_path = directory / "report.json"
209
+ try:
210
+ report = ScoreReport.model_validate_json(report_path.read_text(encoding="utf-8"))
211
+ source = _read_source_tree(directory / "source")
212
+ evaluated = EvaluatedCandidate(candidate_id, source, report)
213
+ except (OSError, ValueError) as error:
214
+ raise ValueError(
215
+ f"run dir {self.run_dir} holds corrupt or missing evidence for "
216
+ f"{candidate_id}: {error}"
217
+ ) from error
218
+ outcomes.append(
219
+ SlotOutcome(
220
+ slot=index,
221
+ candidate_id=candidate_id,
222
+ evaluated=evaluated,
223
+ evidence_dir=str(entry.get("evidence_dir") or ""),
224
+ )
225
+ )
226
+ return tuple(outcomes)
227
+
228
+ def record_pending(self, proposal: CandidateProposal, *, evidence_dir: str = "") -> None:
229
+ """Durably checkpoint one ACCEPTED proposal before its paid evaluation starts.
230
+
231
+ The source tree and proposal record land first; the atomic `pending.json` marker is
232
+ written last, so a marker's presence implies complete candidate bytes. A crash between
233
+ this checkpoint and the boundary commit resumes by rescoring this exact candidate.
234
+ """
235
+ directory = self.candidate_dir(proposal.candidate_id)
236
+ _write_candidate_source(directory / "source", proposal.source)
237
+ write_json_atomic(
238
+ directory / "proposal.json",
239
+ {
240
+ "candidate_id": proposal.candidate_id,
241
+ "doc_hash": proposal.candidate.doc_hash,
242
+ "tree_hash": proposal.source.tree_hash,
243
+ "evidence_dir": evidence_dir,
244
+ },
245
+ )
246
+ write_json_atomic(
247
+ directory / _PENDING_FILE,
248
+ {
249
+ "candidate_id": proposal.candidate_id,
250
+ "doc_hash": proposal.candidate.doc_hash,
251
+ "tree_hash": proposal.source.tree_hash,
252
+ },
253
+ )
254
+
255
+ def load_pending(self, slot: int) -> HarnessSourceTree | None:
256
+ """The checkpointed-but-unscored candidate for `slot`, or None when there is none."""
257
+ candidate_id = candidate_slot_id(slot)
258
+ directory = self.candidate_dir(candidate_id)
259
+ marker_path = directory / _PENDING_FILE
260
+ if not marker_path.is_file():
261
+ return None
262
+ try:
263
+ marker = json.loads(marker_path.read_text(encoding="utf-8"))
264
+ source = _read_source_tree(directory / "source")
265
+ except (OSError, ValueError) as error:
266
+ raise ValueError(
267
+ f"run dir {self.run_dir} holds a corrupt pending candidate {candidate_id}: "
268
+ f"{error}; delete {directory} to redo the proposal"
269
+ ) from error
270
+ if not isinstance(marker, dict) or marker.get("tree_hash") != source.tree_hash:
271
+ raise ValueError(
272
+ f"pending candidate {candidate_id} in {self.run_dir} does not match its "
273
+ f"recorded tree hash; delete {directory} to redo the proposal"
274
+ )
275
+ return source
276
+
277
+ def commit(self, outcomes: Sequence[SlotOutcome]) -> None:
278
+ """Persist the newest outcome's evidence, then atomically commit the ordered index."""
279
+ latest = outcomes[-1]
280
+ directory = self.candidate_dir(latest.candidate_id)
281
+ directory.mkdir(parents=True, exist_ok=True)
282
+ if latest.evaluated is None:
283
+ write_json_atomic(
284
+ directory / "error.json",
285
+ {
286
+ "candidate_id": latest.candidate_id,
287
+ "reason": latest.reason,
288
+ "evidence_dir": latest.evidence_dir,
289
+ },
290
+ )
291
+ else:
292
+ _write_candidate_source(directory / "source", latest.evaluated.source)
293
+ write_json_atomic(
294
+ directory / "report.json", latest.evaluated.report.model_dump(mode="json")
295
+ )
296
+ if latest.slot > 0:
297
+ write_json_atomic(
298
+ directory / "proposal.json",
299
+ {
300
+ "candidate_id": latest.candidate_id,
301
+ "doc_hash": latest.evaluated.report.doc_hash,
302
+ "tree_hash": latest.evaluated.source.tree_hash,
303
+ "evidence_dir": latest.evidence_dir,
304
+ },
305
+ )
306
+ write_json_atomic(
307
+ self.run_dir / _STATE_FILE,
308
+ {"outcomes": [_state_entry(outcome) for outcome in outcomes]},
309
+ )
310
+ # Cleared only AFTER the state commit: a crash in between leaves a stale marker for an
311
+ # already-committed slot, which load_pending never consults again.
312
+ (directory / _PENDING_FILE).unlink(missing_ok=True)
313
+
314
+
315
+ def optimize(
316
+ seed: HarnessSourceTree,
317
+ scorer: Scorer,
318
+ proposer: CandidateProposer,
319
+ iterations: int,
320
+ *,
321
+ run_dir: str | Path,
322
+ should_cancel: Callable[[], bool] | None = None,
323
+ max_new_boundaries: int | None = None,
324
+ on_boundary: Callable[[SlotOutcome], None] | None = None,
325
+ ) -> PopulationResult:
326
+ """Score the seed and `iterations` sequential proposal slots with durable resume.
327
+
328
+ `iterations == 0` is score-only: the seed is scored and committed and no proposal runs,
329
+ which is how a fixed harness (a baseline or a frozen champion) is scored on a task set.
330
+ `max_new_boundaries` stops this invocation after that many NEW boundaries (already
331
+ committed slots do not count), leaving the fixed total plan resumable. The result's
332
+ `completed` flag says whether every slot has been consumed.
333
+ """
334
+ if isinstance(iterations, bool) or not isinstance(iterations, int) or iterations < 0:
335
+ raise ValueError("iterations must be a non-negative integer")
336
+ if max_new_boundaries is not None and (
337
+ isinstance(max_new_boundaries, bool)
338
+ or not isinstance(max_new_boundaries, int)
339
+ or max_new_boundaries < 1
340
+ ):
341
+ raise ValueError("max_new_boundaries must be a positive integer")
342
+ state = PopulationRunState(run_dir)
343
+ outcomes = list(state.load())
344
+ if len(outcomes) > iterations + 1:
345
+ raise ValueError(
346
+ f"run dir {state.run_dir} already holds {len(outcomes)} slot outcomes, more than "
347
+ f"the requested {iterations} iterations allow; rerun with the recorded iterations"
348
+ )
349
+ _validate_resumed(outcomes, seed=seed, scorer=scorer)
350
+ new_boundaries = 0
351
+
352
+ def record(outcome: SlotOutcome) -> None:
353
+ nonlocal new_boundaries
354
+ outcomes.append(outcome)
355
+ state.commit(outcomes)
356
+ new_boundaries += 1
357
+ if on_boundary is not None:
358
+ on_boundary(outcome)
359
+
360
+ _check_cancelled(should_cancel)
361
+ if not outcomes:
362
+ seed_id = candidate_slot_id(0)
363
+ report = scorer.score(seed.to_doc(seed_id), should_cancel=should_cancel)
364
+ record(
365
+ SlotOutcome(
366
+ slot=0,
367
+ candidate_id=seed_id,
368
+ evaluated=EvaluatedCandidate(seed_id, seed, report),
369
+ )
370
+ )
371
+
372
+ while len(outcomes) <= iterations:
373
+ if max_new_boundaries is not None and new_boundaries >= max_new_boundaries:
374
+ break
375
+ _check_cancelled(should_cancel)
376
+ slot = len(outcomes)
377
+ candidate_id = candidate_slot_id(slot)
378
+ # A pending checkpoint means this slot's proposal was already accepted and paid for;
379
+ # skip straight to (re)scoring it instead of buying a fresh proposer turn whose new
380
+ # doc hash would also orphan the part-paid evaluator job.
381
+ source = state.load_pending(slot)
382
+ if source is None:
383
+ population = tuple(
384
+ outcome.evaluated for outcome in outcomes if outcome.evaluated is not None
385
+ )
386
+ try:
387
+ proposal = proposer.propose(population, slot=slot, should_cancel=should_cancel)
388
+ except CandidateProposalError as error:
389
+ logger.info("slot %d consumed by an invalid proposal: %s", slot, error.reason)
390
+ record(
391
+ SlotOutcome(
392
+ slot=slot,
393
+ candidate_id=candidate_id,
394
+ reason=error.reason,
395
+ evidence_dir=error.evidence_dir,
396
+ )
397
+ )
398
+ continue
399
+ if proposal.candidate_id != candidate_id:
400
+ raise ValueError(
401
+ f"proposer returned {proposal.candidate_id!r} for slot {slot}; "
402
+ f"expected {candidate_id!r}"
403
+ )
404
+ state.record_pending(proposal)
405
+ source = proposal.source
406
+ else:
407
+ logger.info("slot %d resumes its checkpointed pending candidate", slot)
408
+ _check_cancelled(should_cancel)
409
+ report = scorer.score(source.to_doc(candidate_id), should_cancel=should_cancel)
410
+ record(
411
+ SlotOutcome(
412
+ slot=slot,
413
+ candidate_id=candidate_id,
414
+ evaluated=EvaluatedCandidate(candidate_id, source, report),
415
+ )
416
+ )
417
+
418
+ population = tuple(outcome.evaluated for outcome in outcomes if outcome.evaluated is not None)
419
+ return PopulationResult(
420
+ outcomes=tuple(outcomes),
421
+ population=population,
422
+ best=_earliest_best(population),
423
+ completed=len(outcomes) == iterations + 1,
424
+ )
425
+
426
+
427
+ def _earliest_best(population: Sequence[EvaluatedCandidate]) -> EvaluatedCandidate:
428
+ """The maximum-score candidate; the EARLIEST one wins ties (seed beats equal successors)."""
429
+ best = population[0]
430
+ for candidate in population[1:]:
431
+ if candidate.score > best.score:
432
+ best = candidate
433
+ return best
434
+
435
+
436
+ def _validate_resumed(
437
+ outcomes: Sequence[SlotOutcome],
438
+ *,
439
+ seed: HarnessSourceTree,
440
+ scorer: Scorer,
441
+ ) -> None:
442
+ if not outcomes:
443
+ return
444
+ recorded_seed = outcomes[0].evaluated
445
+ if recorded_seed is None:
446
+ raise ValueError("run dir state is corrupt: the seed slot is recorded as invalid")
447
+ if recorded_seed.source.tree_hash != seed.tree_hash:
448
+ raise ValueError(
449
+ "run dir belongs to a different seed source tree; start a fresh run dir for a new seed"
450
+ )
451
+ for outcome in outcomes:
452
+ if outcome.evaluated is not None and outcome.evaluated.report.request != scorer.request:
453
+ raise ValueError(
454
+ "run dir was scored under a different task-by-attempt request; "
455
+ "rerun with the recorded tasks and attempts or start a fresh run dir"
456
+ )
457
+
458
+
459
+ def _state_entry(outcome: SlotOutcome) -> dict[str, JsonValue]:
460
+ if outcome.evaluated is None:
461
+ return {
462
+ "slot": outcome.slot,
463
+ "candidate_id": outcome.candidate_id,
464
+ "kind": "invalid",
465
+ "reason": outcome.reason,
466
+ "evidence_dir": outcome.evidence_dir,
467
+ }
468
+ return {
469
+ "slot": outcome.slot,
470
+ "candidate_id": outcome.candidate_id,
471
+ "kind": "scored",
472
+ "score": outcome.evaluated.score,
473
+ "doc_hash": outcome.evaluated.report.doc_hash,
474
+ "evidence_dir": outcome.evidence_dir,
475
+ }
476
+
477
+
478
+ def _write_candidate_source(directory: Path, source: HarnessSourceTree) -> None:
479
+ """Replace one candidate's source dir with exact bytes (no leftovers, no newline drift).
480
+
481
+ Clearing first matters: a crashed prior attempt's leftover files would otherwise merge into
482
+ a redone slot's tree and fail doc-hash re-verification on every later load. Bytes are
483
+ written and read without newline translation so content hashes round-trip exactly.
484
+ """
485
+ if directory.exists():
486
+ shutil.rmtree(directory)
487
+ for item in source.files:
488
+ target = directory / item.path
489
+ target.parent.mkdir(parents=True, exist_ok=True)
490
+ target.write_bytes(item.content.encode("utf-8"))
491
+
492
+
493
+ def _read_source_tree(directory: Path) -> HarnessSourceTree:
494
+ # read_bytes + exact decode: Path.read_text's universal newlines would fold \r into \n and
495
+ # silently change content hashes on every resume.
496
+ files = [
497
+ HarnessSourceFile(
498
+ path=path.relative_to(directory).as_posix(),
499
+ content=path.read_bytes().decode("utf-8"),
500
+ )
501
+ for path in sorted(directory.rglob("*"))
502
+ if path.is_file()
503
+ ]
504
+ return HarnessSourceTree(files=tuple(files))
505
+
506
+
507
+ def _check_cancelled(should_cancel: Callable[[], bool] | None) -> None:
508
+ if should_cancel is not None and should_cancel():
509
+ raise HarnessSearchCancelled("harness search cancelled")