world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,288 @@
1
+ """`CodeRuntime`: the agent loop itself as a searchable harness surface.
2
+
3
+ Two live search campaigns showed the same wall: the meta-agent diagnoses failure mechanisms
4
+ correctly, but prompt- and skill-level edits cannot express the fixes that actually make a
5
+ harness good — loop structure, retries, context compaction, observation truncation, token
6
+ budgets. Those are *code*. A `code:runtime` surface holds a Python module defining
7
+ `run(kit) -> str`; the search edits that program.
8
+
9
+ The contract is the `RuntimeKit`, and it carries three guarantees the search relies on:
10
+
11
+ - **Budgeted.** `kit.complete` and `kit.execute` enforce hard caps on LLM calls and environment
12
+ actions. A runaway loop raises `BudgetExceeded` instead of wedging an eval; cost is bounded per
13
+ episode by construction, not by hope.
14
+ - **Kit-recorded.** Every environment action is appended to the transcript by the kit itself, so
15
+ the judge always scores ground truth: generated code cannot fake, omit, or reorder what it did.
16
+ - **Crash-isolated.** An exception inside `run` fails that episode (scored as a failure with the
17
+ partial transcript), never the evaluation loop around it.
18
+
19
+ The kit is an interface contract, not a security boundary: harness code runs in-process and is
20
+ trusted to the same degree as the rest of the search (it is reviewed via the delta archive, and
21
+ only ever exercised against the world model during search). Running searched code against real
22
+ environments is a deployment decision that belongs behind a sandbox.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ from collections.abc import Callable
28
+ from typing import cast
29
+
30
+ from pydantic import BaseModel, Field
31
+
32
+ from wmo.core.types import Action, ActionKind, EnvState, JsonObject, Observation, Step
33
+ from wmo.harness.environment import AgentEnvironment, is_env_action
34
+ from wmo.harness.runtime import RunResult, StopReason
35
+ from wmo.harness.skills import SkillLibrary
36
+ from wmo.harness.tools import ToolCall, ToolSpec, parse_tool_call, render_tools
37
+ from wmo.providers.base import Message, Provider
38
+
39
+ CODE_ENTRYPOINT = "run"
40
+
41
+ # Defaults sized so a reasonable loop never notices them: the fixed baseline loop uses at most
42
+ # `max_turns` (20) of each.
43
+ DEFAULT_MAX_LLM_CALLS = 40
44
+ DEFAULT_MAX_ENV_ACTIONS = 40
45
+
46
+
47
+ class RunBudget(BaseModel):
48
+ """Hard per-episode caps enforced by the kit."""
49
+
50
+ max_llm_calls: int = Field(default=DEFAULT_MAX_LLM_CALLS, ge=1)
51
+ max_env_actions: int = Field(default=DEFAULT_MAX_ENV_ACTIONS, ge=1)
52
+
53
+
54
+ class BudgetExceeded(RuntimeError):
55
+ """Raised by the kit when harness code exhausts an episode budget."""
56
+
57
+
58
+ class RuntimeKit:
59
+ """Everything harness code may touch, and the recorder of what it actually did."""
60
+
61
+ def __init__(
62
+ self,
63
+ *,
64
+ task_id: str,
65
+ instruction: str,
66
+ environment: AgentEnvironment,
67
+ provider: Provider,
68
+ tools: list[ToolSpec],
69
+ skills: SkillLibrary,
70
+ temperature: float,
71
+ budget: RunBudget,
72
+ system_prompt: str = "",
73
+ ) -> None:
74
+ self.task_id = task_id
75
+ self.instruction = instruction
76
+ self.temperature = temperature
77
+ self.tools = tools
78
+ # The doc's assembled prompt (prompt surfaces + tools + skills index): prompt surfaces
79
+ # stay meaningful alongside a code surface, and code may use or ignore this.
80
+ self.system_prompt = system_prompt
81
+ self._environment = environment
82
+ self._provider = provider
83
+ self._skills = skills
84
+ self._budget = budget
85
+ self._llm_calls = 0
86
+ self._env_actions = 0
87
+ self.steps: list[Step] = []
88
+
89
+ # -- the two budgeted primitives -----------------------------------------------------------
90
+
91
+ def complete(
92
+ self,
93
+ system: str,
94
+ messages: list[Message] | list[tuple[str, str]],
95
+ *,
96
+ temperature: float | None = None,
97
+ max_tokens: int = 2048,
98
+ ) -> str:
99
+ """One LLM call. Accepts `Message`s or plain `(role, content)` tuples."""
100
+ if self._llm_calls >= self._budget.max_llm_calls:
101
+ raise BudgetExceeded(f"llm call budget exhausted ({self._budget.max_llm_calls})")
102
+ self._llm_calls += 1
103
+ normalized = [_to_message(m) for m in messages]
104
+ completion = self._provider.complete(
105
+ system,
106
+ normalized,
107
+ temperature=self.temperature if temperature is None else temperature,
108
+ max_tokens=max_tokens,
109
+ )
110
+ return completion.text
111
+
112
+ def execute(self, tool: str, arguments: JsonObject) -> Observation:
113
+ """One environment action, validated against the tool policy and recorded verbatim."""
114
+ if self._env_actions >= self._budget.max_env_actions:
115
+ raise BudgetExceeded(f"env action budget exhausted ({self._budget.max_env_actions})")
116
+ action = Action(kind=ActionKind.TOOL_CALL, name=tool, arguments=arguments)
117
+ if tool not in {t.name for t in self.tools} or not is_env_action(action):
118
+ observation = Observation(content=f"tool {tool!r} not available", is_error=True)
119
+ else:
120
+ self._env_actions += 1
121
+ observation = self._environment.execute(action)
122
+ self.steps.append(
123
+ Step(
124
+ action=action,
125
+ observation=observation,
126
+ state_before=EnvState(),
127
+ task=self.instruction,
128
+ )
129
+ )
130
+ return observation
131
+
132
+ # -- conveniences (unbudgeted, side-effect free) --------------------------------------------
133
+
134
+ def parse_tool_call(self, text: str) -> ToolCall | None:
135
+ return parse_tool_call(text)
136
+
137
+ def tools_text(self) -> str:
138
+ return render_tools(self.tools)
139
+
140
+ def skills_index(self) -> str:
141
+ return self._skills.render_index()
142
+
143
+ def read_skill(self, name: str) -> str | None:
144
+ skill = self._skills.get(name)
145
+ return skill.body if skill is not None else None
146
+
147
+
148
+ def _to_message(m: Message | tuple[str, str]) -> Message:
149
+ if isinstance(m, Message):
150
+ return m
151
+ role, content = m
152
+ if role == "user":
153
+ return Message(role="user", content=content)
154
+ if role == "assistant":
155
+ return Message(role="assistant", content=content)
156
+ raise ValueError(f"message role must be 'user' or 'assistant', got {role!r}")
157
+
158
+
159
+ def compile_harness_code(code: str) -> None:
160
+ """Front-loaded validation: the code must compile and define `run` at module scope.
161
+
162
+ Raises `ValueError` so `HarnessDoc` construction rejects an unrunnable harness before any
163
+ eval budget could be spent on it. Behavioral quality is the gate's job, not this one's.
164
+ """
165
+ try:
166
+ compiled = compile(code, "<code:runtime>", "exec")
167
+ except SyntaxError as exc:
168
+ raise ValueError(f"code:runtime does not compile: {exc}") from exc
169
+ names = set(compiled.co_names)
170
+ # A module that never binds `run` can't be dispatched. co_names covers references, so also
171
+ # accept a top-level def by scanning consts for the code object.
172
+ defines_run = (
173
+ any(getattr(const, "co_name", None) == CODE_ENTRYPOINT for const in compiled.co_consts)
174
+ or CODE_ENTRYPOINT in names
175
+ )
176
+ if not defines_run:
177
+ raise ValueError(f"code:runtime must define `{CODE_ENTRYPOINT}(kit)` at module scope")
178
+
179
+
180
+ class CodeRuntime:
181
+ """Drives episodes through a harness-defined `run(kit)` instead of the fixed loop."""
182
+
183
+ def __init__(
184
+ self,
185
+ provider: Provider,
186
+ *,
187
+ code: str,
188
+ tools: list[ToolSpec],
189
+ temperature: float = 0.7,
190
+ skills: SkillLibrary | None = None,
191
+ budget: RunBudget | None = None,
192
+ system_prompt: str = "",
193
+ ) -> None:
194
+ compile_harness_code(code)
195
+ namespace: dict[str, object] = {"__name__": "wmo_harness_code"}
196
+ exec(code, namespace) # noqa: S102 - the code surface IS the artifact under search
197
+ entry = namespace.get(CODE_ENTRYPOINT)
198
+ if not callable(entry):
199
+ raise ValueError(f"code:runtime `{CODE_ENTRYPOINT}` is not callable")
200
+ self._run = cast("Callable[[RuntimeKit], object]", entry)
201
+ self._provider = provider
202
+ self._tools = tools
203
+ self._temperature = temperature
204
+ self._skills = skills if skills is not None else SkillLibrary()
205
+ self._budget = budget if budget is not None else RunBudget()
206
+ self._system_prompt = system_prompt
207
+
208
+ def run(self, task_id: str, instruction: str, environment: AgentEnvironment) -> RunResult:
209
+ kit = RuntimeKit(
210
+ task_id=task_id,
211
+ instruction=instruction,
212
+ environment=environment,
213
+ provider=self._provider,
214
+ tools=self._tools,
215
+ skills=self._skills,
216
+ temperature=self._temperature,
217
+ budget=self._budget,
218
+ system_prompt=self._system_prompt,
219
+ )
220
+ try:
221
+ answer = self._run(kit)
222
+ except BudgetExceeded as exc:
223
+ return self._result(kit, StopReason.BUDGET, note=str(exc))
224
+ except Exception as exc: # noqa: BLE001 - crash-isolation is the contract
225
+ return self._result(kit, StopReason.ERROR, note=f"{type(exc).__name__}: {exc}")
226
+ return RunResult(
227
+ task_id=kit.task_id,
228
+ steps=kit.steps,
229
+ stop_reason=StopReason.SUBMITTED,
230
+ answer=answer if isinstance(answer, str) else "",
231
+ turns=len(kit.steps),
232
+ )
233
+
234
+ def _result(self, kit: RuntimeKit, stop_reason: StopReason, *, note: str) -> RunResult:
235
+ # The kit-recorded partial transcript survives, plus one error step so the judge (and the
236
+ # failure clustering) can see WHY the episode ended.
237
+ kit.steps.append(
238
+ Step(
239
+ action=Action(kind=ActionKind.MESSAGE, content="(harness runtime)"),
240
+ observation=Observation(content=note, is_error=True),
241
+ state_before=EnvState(),
242
+ task=kit.instruction,
243
+ )
244
+ )
245
+ return RunResult(
246
+ task_id=kit.task_id,
247
+ steps=kit.steps,
248
+ stop_reason=stop_reason,
249
+ answer="",
250
+ turns=len(kit.steps),
251
+ )
252
+
253
+
254
+ # The reference loop, as the seed content of a `code:runtime` surface: functionally the fixed
255
+ # `AgentRuntime` loop, expressed through the kit so the search can restructure it.
256
+ DEFAULT_RUNTIME_CODE = '''"""Baseline agent loop: one JSON tool call per turn.
257
+
258
+ Calling `submit` ends the episode."""
259
+
260
+
261
+ def run(kit):
262
+ messages = [("user", "TASK: " + kit.instruction)]
263
+ nudged = False
264
+ for _ in range(20):
265
+ reply = kit.complete(kit.system_prompt, messages).strip()
266
+ call = kit.parse_tool_call(reply)
267
+ if call is None:
268
+ if nudged:
269
+ return ""
270
+ nudged = True
271
+ messages.append(("assistant", reply))
272
+ messages.append(("user", "[ERROR] reply with EXACTLY one JSON object: "
273
+ "{\\"tool\\": \\"<name>\\", \\"arguments\\": {...}}"))
274
+ continue
275
+ if call.tool == "submit":
276
+ answer = call.arguments.get("answer")
277
+ return answer if isinstance(answer, str) else ""
278
+ if call.tool == "read_skill":
279
+ name = call.arguments.get("name")
280
+ body = kit.read_skill(name if isinstance(name, str) else "")
281
+ text, is_error = (body, False) if body is not None else ("no such skill", True)
282
+ else:
283
+ observation = kit.execute(call.tool, call.arguments)
284
+ text, is_error = observation.content, observation.is_error
285
+ messages.append(("assistant", reply))
286
+ messages.append(("user", ("[ERROR] " if is_error else "[OK] ") + text))
287
+ return ""
288
+ '''