world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
wmo/cli/e2b_cmds.py ADDED
@@ -0,0 +1,212 @@
1
+ """`wmo e2b reap`: reclaim E2B sandbox slots held by orphaned harbor trial sandboxes.
2
+
3
+ The operator-facing half of `wmo.harness.e2b_reap`. It renders the reap candidates and their
4
+ evidence, and it never kills anything without `--yes`: a dry run is the default because every
5
+ kill is destructive and one wrong id is somebody's running trial.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from datetime import UTC, datetime
11
+
12
+ import typer
13
+ from rich.console import Console
14
+ from rich.table import Table
15
+
16
+ from wmo.harness.e2b_ledger import read_ledger_files
17
+ from wmo.harness.e2b_reap import (
18
+ AliveSandbox,
19
+ ReapCandidate,
20
+ execute_reap,
21
+ kill_sandbox,
22
+ list_alive_sandboxes,
23
+ plan_reap,
24
+ sandbox_cap,
25
+ )
26
+
27
+ _console = Console()
28
+
29
+
30
+ def _now() -> datetime:
31
+ """The instant sandbox ages are measured against (a seam tests pin)."""
32
+ return datetime.now(UTC)
33
+
34
+
35
+ e2b_app = typer.Typer(help="Inspect and reclaim E2B sandbox capacity.", no_args_is_help=True)
36
+
37
+ # Module-level singletons: a typer.Option call cannot be a default inline (ruff B008).
38
+ _REAP_YES = typer.Option(
39
+ False, "--yes", help="Actually kill the candidates. Without it this is a dry run."
40
+ )
41
+ _REAP_DEAD_OWNERS = typer.Option(
42
+ True,
43
+ "--dead-owners/--no-dead-owners",
44
+ help="Include sandboxes recorded by a wmo run on THIS machine whose process is gone "
45
+ "(exact ids from the local ledger; safe and on by default).",
46
+ )
47
+ _REAP_STALE_MINUTES = typer.Option(
48
+ None,
49
+ "--stale-minutes",
50
+ min=1,
51
+ help="Also include any harbor trial sandbox on the ACCOUNT started more than N minutes "
52
+ "ago. This match is account-wide and can kill another machine's live run, so it is "
53
+ "opt-in; sandboxes owned by a live local process are still excluded.",
54
+ )
55
+
56
+
57
+ @e2b_app.command("reap")
58
+ def reap(
59
+ yes: bool = _REAP_YES,
60
+ dead_owners: bool = _REAP_DEAD_OWNERS,
61
+ stale_minutes: int | None = _REAP_STALE_MINUTES,
62
+ ) -> None:
63
+ """Free E2B concurrency slots held by sandboxes no run is using any more.
64
+
65
+ A harbor trial sandbox keeps its slot until its own multi-hour timeout, so a run that dies
66
+ without graceful shutdown (crash, SIGKILL, budget abort, machine sleep) leaves orphans that
67
+ starve every later run at the account cap of 100 concurrent sandboxes (override with
68
+ `$WMO_E2B_SANDBOX_CAP`).
69
+
70
+ Two evidence classes, most conservative first:
71
+
72
+ - `--dead-owners` (default): unreleased entries in this machine's sandbox ledger whose
73
+ owning process is gone. These are provably orphans of local runs and are killed by exact
74
+ recorded id.
75
+ - `--stale-minutes N` (opt-in): every sandbox ON THE ACCOUNT that carries harbor trial
76
+ metadata and started more than N minutes ago. The match is account-wide, so it can kill
77
+ a run on another machine or in another checkout; use it only when you know no such run
78
+ should be alive. Sandboxes whose local owner process is still running are never selected.
79
+
80
+ The default is a dry run that prints the candidates and changes nothing. Pass `--yes` to
81
+ kill them.
82
+ """
83
+ try:
84
+ cap = sandbox_cap()
85
+ except ValueError as error:
86
+ raise typer.BadParameter(str(error)) from error
87
+ if not dead_owners and stale_minutes is None:
88
+ raise typer.BadParameter(
89
+ "--no-dead-owners with no --stale-minutes selects nothing; drop --no-dead-owners "
90
+ "or add --stale-minutes N"
91
+ )
92
+ try:
93
+ alive = list_alive_sandboxes()
94
+ except ImportError as error:
95
+ raise typer.BadParameter(str(error)) from error
96
+ except Exception as error: # noqa: BLE001 - any provider failure is a usage-level message
97
+ raise typer.BadParameter(
98
+ f"could not list E2B sandboxes ({type(error).__name__}: {error}); check "
99
+ "$E2B_API_KEY and your connection"
100
+ ) from error
101
+
102
+ plan = plan_reap(
103
+ alive=alive,
104
+ ledger_files=read_ledger_files(),
105
+ now=_now(),
106
+ dead_owners=dead_owners,
107
+ stale_minutes=stale_minutes,
108
+ )
109
+ _print_usage(alive, cap)
110
+ if not plan.candidates:
111
+ _console.print("[green]nothing to reap[/green]: no orphaned sandbox matched")
112
+ if not plan.vanished:
113
+ return
114
+ if not yes:
115
+ _console.print(
116
+ f"{len(plan.vanished)} ledger record(s) name sandboxes E2B no longer has; "
117
+ "--yes clears them"
118
+ )
119
+ return
120
+ outcome = execute_reap(plan, killer=kill_sandbox)
121
+ _console.print(
122
+ f"released {len(plan.vanished)} stale ledger record(s); "
123
+ f"pruned {len(outcome.pruned_ledgers)} ledger file(s)"
124
+ )
125
+ return
126
+
127
+ _console.print(_candidates_table(plan.candidates))
128
+ if not yes:
129
+ _console.print(
130
+ f"[yellow]dry run[/yellow]: {len(plan.candidates)} sandbox(es) would be killed, "
131
+ f"freeing up to {len(plan.candidates)} of {cap} slot(s). Re-run with --yes to do it."
132
+ )
133
+ return
134
+
135
+ outcome = execute_reap(plan, killer=kill_sandbox)
136
+ remaining = max(len(alive) - outcome.freed, 0)
137
+ _console.print(
138
+ f"[green]reaped[/green] {outcome.freed} sandbox(es); usage now {remaining}/{cap} "
139
+ f"({max(cap - remaining, 0)} free)"
140
+ )
141
+ if outcome.already_gone:
142
+ _console.print(
143
+ f"{len(outcome.already_gone)} candidate(s) were already gone: "
144
+ f"{', '.join(outcome.already_gone)}"
145
+ )
146
+ for sandbox_id, error in outcome.failed:
147
+ _console.print(f"[red]kill failed[/red] {sandbox_id}: {error}")
148
+ if outcome.pruned_ledgers:
149
+ _console.print(f"pruned {len(outcome.pruned_ledgers)} fully released ledger file(s)")
150
+
151
+
152
+ def _print_usage(alive: list[AliveSandbox], cap: int) -> None:
153
+ """Print current account usage against the cap."""
154
+ trials = sum(1 for sandbox in alive if sandbox.is_harbor_trial())
155
+ _console.print(
156
+ f"E2B usage: [bold]{len(alive)}/{cap}[/bold] concurrent sandbox(es) running "
157
+ f"({trials} harbor trial environment(s)), {max(cap - len(alive), 0)} slot(s) free"
158
+ )
159
+
160
+
161
+ def _candidates_table(candidates: tuple[ReapCandidate, ...]) -> Table:
162
+ """Render one row per reap candidate with the evidence behind it."""
163
+ table = Table(title="Reap candidates")
164
+ table.add_column("Sandbox", no_wrap=True)
165
+ table.add_column("Age", justify="right", no_wrap=True)
166
+ table.add_column("Template", no_wrap=True)
167
+ table.add_column("Source", no_wrap=True)
168
+ table.add_column("Owner", justify="right", no_wrap=True)
169
+ table.add_column("Alive", no_wrap=True)
170
+ table.add_column("Trial")
171
+ for candidate in candidates:
172
+ table.add_row(
173
+ candidate.sandbox_id,
174
+ _age(candidate.age_seconds),
175
+ _template(candidate.template_id),
176
+ candidate.source,
177
+ "unknown" if candidate.owner_pid is None else str(candidate.owner_pid),
178
+ _owner_liveness(candidate.owner_alive),
179
+ candidate.trial_name or "",
180
+ )
181
+ return table
182
+
183
+
184
+ # A wmo harbor alias is `wmo-hb-v1-<64 hex>`: printing it whole squeezes every evidence column
185
+ # out of the table, and its head plus tail already identify a template uniquely in practice.
186
+ _TEMPLATE_DISPLAY_LEN = 22
187
+
188
+
189
+ def _template(template_id: str) -> str:
190
+ """Shorten a long template alias, keeping both ends so two aliases stay distinguishable."""
191
+ if len(template_id) <= _TEMPLATE_DISPLAY_LEN:
192
+ return template_id
193
+ return f"{template_id[: _TEMPLATE_DISPLAY_LEN - 8]}…{template_id[-7:]}"
194
+
195
+
196
+ def _owner_liveness(owner_alive: bool | None) -> str:
197
+ """Render owner liveness, distinguishing "no recorded owner" from "owner is gone"."""
198
+ if owner_alive is None:
199
+ return "unknown"
200
+ return "yes" if owner_alive else "no"
201
+
202
+
203
+ def _age(seconds: float) -> str:
204
+ """Human-readable sandbox age, e.g. `3h12m`."""
205
+ minutes = int(seconds // 60)
206
+ hours, minutes = divmod(minutes, 60)
207
+ return f"{hours}h{minutes:02d}m" if hours else f"{minutes}m"
208
+
209
+
210
+ def register(app: typer.Typer) -> None:
211
+ """Attach the `wmo e2b` command group to the root CLI."""
212
+ app.add_typer(e2b_app, name="e2b")
@@ -0,0 +1,207 @@
1
+ """`wmo eval --mode closed-loop` and `wmo eval agreement` — the closed-loop halves of eval.
2
+
3
+ Kept out of `app.py` so the (large) eval command stays readable; `app.py` routes here.
4
+ Closed-loop mode runs an agent harness against a built world model — the environment is ALWAYS
5
+ the world-model simulation; `--harness-backend e2b` only moves the pi-node harness PROCESS into
6
+ pooled E2B sandboxes (its tool calls stay answered host-side) — and scores task success;
7
+ `agreement` compares two saved closed-loop reports — the outcome-agreement check
8
+ docs/reference/closed_loop.md names.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from pathlib import Path
14
+
15
+ import typer
16
+ from rich.console import Console
17
+
18
+ from wmo.cli.model_roles import resolve_opt_in_model_provider
19
+ from wmo.config import WorldModelStore
20
+ from wmo.engine import load_world_model
21
+ from wmo.evals.agreement import compute_agreement
22
+ from wmo.evals.closed_loop import ClosedLoopEval, ClosedLoopReport
23
+ from wmo.evals.gold import GoldJudge, GoldVerdict
24
+ from wmo.evals.tasks import load_tasks
25
+ from wmo.harness.doc import MAX_TURNS_ID, HarnessDoc, Surface, SurfaceKind
26
+ from wmo.harness.runtime import DEFAULT_MAX_TURNS, AgentRuntime
27
+ from wmo.harness.store import HarnessStore
28
+
29
+
30
+ def run_closed_loop(
31
+ console: Console,
32
+ *,
33
+ tasks_file: str,
34
+ name: str | None,
35
+ root: str,
36
+ k: int,
37
+ max_turns: int | None,
38
+ out: str | None,
39
+ harness: str | None = None,
40
+ harness_backend: str = "local",
41
+ eval_concurrency: int | None = None,
42
+ e2b_template: str | None = None,
43
+ ) -> None:
44
+ """Run an agent harness on each task against the world model; print and optionally save.
45
+
46
+ `--harness <name>[@ref]` runs a stored harness version (ref = version or alias; default is
47
+ the champion alias); without it the built-in baseline loop runs. `max_turns=None` means "the
48
+ harness's own cap" (or the default for the baseline); an explicit value overrides either —
49
+ never silently ignored. `[models.agent]` selects the agent model for either path; when unset,
50
+ the agent shares the world model's provider. The environment is ALWAYS the world-model
51
+ simulation;
52
+ `--harness-backend e2b` only moves the pi-node harness PROCESS into pooled E2B sandboxes
53
+ (tool calls stay answered by the world model host-side), running all (task, attempt) cells
54
+ at once unless `--eval-concurrency` caps them.
55
+ """
56
+ if harness_backend not in ("local", "e2b"):
57
+ raise typer.BadParameter(
58
+ f"unknown --harness-backend {harness_backend!r}; choose local or e2b"
59
+ )
60
+ try:
61
+ tasks = load_tasks(tasks_file)
62
+ except (OSError, ValueError) as exc: # missing file, malformed JSONL, empty, duplicate ids
63
+ raise typer.BadParameter(f"cannot load tasks from {tasks_file!r}: {exc}") from exc
64
+ # The world model IS the environment on every backend, so it is always required.
65
+ store = WorldModelStore(root)
66
+ try:
67
+ model_dir = store.resolve(name)
68
+ except (FileNotFoundError, ValueError) as exc:
69
+ raise typer.BadParameter(str(exc)) from exc
70
+ world_model, provider = load_world_model(model_dir)
71
+ agent_provider, agent_model = resolve_opt_in_model_provider(root, "agent", provider)
72
+
73
+ loaded_harness = _load_harness(harness, root)
74
+ if loaded_harness is None and harness_backend == "e2b":
75
+ raise typer.BadParameter(
76
+ "--harness-backend e2b runs a pi-node harness process in sandboxes; the built-in "
77
+ "baseline loop has no such process — pass --harness"
78
+ )
79
+ agent_label = (
80
+ f"{loaded_harness.name}-v{loaded_harness.version}"
81
+ if loaded_harness is not None
82
+ else "baseline"
83
+ )
84
+ agent_identity = (
85
+ f"{agent_provider.config.kind.value}:{agent_model}" if agent_model is not None else None
86
+ )
87
+ report_agent_label = (
88
+ f"{agent_label}[agent={agent_identity}]" if agent_identity is not None else agent_label
89
+ )
90
+ agent_note = (
91
+ f" using agent model [bold]{agent_identity}[/bold]" if agent_identity is not None else ""
92
+ )
93
+ versus = (
94
+ f"world model [bold]{model_dir.name}[/bold]"
95
+ if harness_backend == "local"
96
+ else f"world model [bold]{model_dir.name}[/bold] (pi harness in pooled E2B sandboxes)"
97
+ )
98
+ console.print(
99
+ f"closed-loop: harness [bold]{agent_label}[/bold]{agent_note} vs {versus} "
100
+ f"on {len(tasks)} task(s), k={k}…"
101
+ )
102
+
103
+ def _progress(task_id: str, attempt: int, verdict: GoldVerdict) -> None:
104
+ mark = "[green]pass[/green]" if verdict.passed else "[red]fail[/red]"
105
+ console.print(f" {task_id} #{attempt}: {mark} ({verdict.rationale})")
106
+
107
+ if loaded_harness is not None:
108
+ if (
109
+ harness_backend == "local"
110
+ and loaded_harness.runtime_kind() == "pi-node"
111
+ and eval_concurrency is not None
112
+ and eval_concurrency != 1
113
+ ):
114
+ # Local pi runtimes are single-episode resources (one runner port/workdir, or one
115
+ # RunnerLink channel): parallel cells would collide.
116
+ raise typer.BadParameter(
117
+ "pi-node harnesses run one episode at a time under --harness-backend local; "
118
+ "drop --eval-concurrency or use --harness-backend e2b"
119
+ )
120
+ if max_turns is not None and max_turns != loaded_harness.max_turns():
121
+ console.print(
122
+ f" note: --max-turns {max_turns} overrides the harness's own "
123
+ f"max_turns={loaded_harness.max_turns()}"
124
+ )
125
+ loaded_harness = _with_max_turns(loaded_harness, max_turns)
126
+ try:
127
+ runtime = loaded_harness.runtime(
128
+ agent_provider,
129
+ backend=harness_backend,
130
+ e2b_template=e2b_template,
131
+ )
132
+ except ValueError as exc: # e.g. e2b on a non-pi-node harness -> usage error
133
+ raise typer.BadParameter(str(exc)) from exc
134
+ else:
135
+ runtime = AgentRuntime(agent_provider, max_turns=max_turns or DEFAULT_MAX_TURNS)
136
+ try:
137
+ evaluation = ClosedLoopEval(
138
+ tasks,
139
+ world_model,
140
+ agent_provider,
141
+ GoldJudge(provider),
142
+ label=f"{report_agent_label}@{model_dir.name}",
143
+ k=k,
144
+ concurrency=(
145
+ eval_concurrency
146
+ if eval_concurrency is not None
147
+ else (0 if harness_backend == "e2b" else 1)
148
+ ),
149
+ runtime=runtime,
150
+ on_progress=_progress,
151
+ )
152
+ report = evaluation.run()
153
+ finally:
154
+ if harness_backend == "e2b":
155
+ # An eval-owned e2b runtime owns a private sandbox pool; tear it down with the eval.
156
+ from wmo.harness.pi_e2b import E2BPiRuntime
157
+
158
+ if isinstance(runtime, E2BPiRuntime):
159
+ runtime.close()
160
+ for task_id, outcome in report.per_task.items():
161
+ console.print(
162
+ f" {task_id:24} success={outcome.success_rate:.2f} "
163
+ f"assertions={outcome.mean_fraction:.2f}"
164
+ )
165
+ console.print(f"[bold]OVERALL[/bold] {report.summary()}")
166
+ if out:
167
+ Path(out).write_text(report.model_dump_json(indent=2), encoding="utf-8")
168
+ console.print(f"wrote closed-loop report -> {out}")
169
+
170
+
171
+ def run_agreement(console: Console, *, report_a: str, report_b: str, threshold: float) -> None:
172
+ """Compare two saved closed-loop reports task-by-task and print the agreement verdict."""
173
+ a = _load_report(report_a)
174
+ b = _load_report(report_b)
175
+ result = compute_agreement(a, b, pass_threshold=threshold)
176
+ c = result.confusion
177
+ la, lb = result.label_a or "A", result.label_b or "B"
178
+ console.print(f"[bold]task verdict confusion[/bold] ({la} vs {lb}):")
179
+ console.print(f" {la}-pass & {lb}-pass: {c.a_pass_b_pass}")
180
+ console.print(f" {la}-pass & {lb}-FAIL: {c.a_pass_b_fail} (A over-credits these)")
181
+ console.print(f" {la}-FAIL & {lb}-pass: {c.a_fail_b_pass}")
182
+ console.print(f" {la}-FAIL & {lb}-FAIL: {c.a_fail_b_fail}")
183
+ console.print(f"[bold]VERDICT[/bold] {result.summary()}")
184
+
185
+
186
+ def _load_report(path: str) -> ClosedLoopReport:
187
+ try:
188
+ return ClosedLoopReport.model_validate_json(Path(path).read_text(encoding="utf-8"))
189
+ except (OSError, ValueError) as exc:
190
+ raise typer.BadParameter(f"cannot read closed-loop report {path!r}: {exc}") from exc
191
+
192
+
193
+ def _with_max_turns(doc: HarnessDoc, max_turns: int) -> HarnessDoc:
194
+ """A copy of `doc` with its max-turns surface replaced (re-validated via the constructor)."""
195
+ surfaces = [s for s in doc.surfaces if s.id != MAX_TURNS_ID]
196
+ surfaces.append(Surface(id=MAX_TURNS_ID, kind=SurfaceKind.PARAM, content=str(max_turns)))
197
+ return HarnessDoc(name=doc.name, version=doc.version, surfaces=surfaces)
198
+
199
+
200
+ def _load_harness(name: str | None, root: str) -> HarnessDoc | None:
201
+ if name is None:
202
+ return None
203
+ base, _, ref = name.partition("@")
204
+ try:
205
+ return HarnessStore(root).load(base, ref or None)
206
+ except (FileNotFoundError, ValueError) as exc:
207
+ raise typer.BadParameter(str(exc)) from exc