world-model-optimizer 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_waterfall/LICENSE +21 -0
- llm_waterfall/__init__.py +53 -0
- llm_waterfall/adapters/__init__.py +36 -0
- llm_waterfall/adapters/anthropic.py +105 -0
- llm_waterfall/adapters/aws_mantle.py +47 -0
- llm_waterfall/adapters/azure_openai.py +71 -0
- llm_waterfall/adapters/base.py +51 -0
- llm_waterfall/adapters/bedrock.py +309 -0
- llm_waterfall/adapters/openai.py +130 -0
- llm_waterfall/classify.py +184 -0
- llm_waterfall/pricing.py +110 -0
- llm_waterfall/py.typed +0 -0
- llm_waterfall/types.py +295 -0
- llm_waterfall/waterfall.py +255 -0
- wmo/__init__.py +38 -0
- wmo/agents/__init__.py +7 -0
- wmo/agents/default.py +29 -0
- wmo/agents/meta.py +55 -0
- wmo/agents/optimizer.py +55 -0
- wmo/agents/project.py +928 -0
- wmo/cli/__init__.py +5 -0
- wmo/cli/agent_session.py +1123 -0
- wmo/cli/app.py +2489 -0
- wmo/cli/e2b_cmds.py +212 -0
- wmo/cli/eval_closed_loop.py +207 -0
- wmo/cli/harness_app.py +1147 -0
- wmo/cli/harness_distill.py +659 -0
- wmo/cli/hosted_session.py +880 -0
- wmo/cli/ingest_cmd.py +165 -0
- wmo/cli/model_roles.py +82 -0
- wmo/cli/platform_cmds.py +372 -0
- wmo/cli/route_app.py +274 -0
- wmo/cli/session_state.py +243 -0
- wmo/cli/ui.py +1107 -0
- wmo/cli/workspace_sync.py +504 -0
- wmo/config/__init__.py +60 -0
- wmo/config/card.py +129 -0
- wmo/config/config.py +367 -0
- wmo/config/dotenv.py +67 -0
- wmo/config/settings.py +128 -0
- wmo/config/store.py +177 -0
- wmo/conftest.py +19 -0
- wmo/connect/__init__.py +88 -0
- wmo/connect/apps.py +78 -0
- wmo/connect/brave.py +284 -0
- wmo/connect/connector.py +79 -0
- wmo/connect/credentials.py +164 -0
- wmo/connect/github.py +321 -0
- wmo/connect/google.py +627 -0
- wmo/connect/notion.py +790 -0
- wmo/connect/oauth.py +461 -0
- wmo/connect/slack.py +555 -0
- wmo/connect/store.py +199 -0
- wmo/connect/types.py +156 -0
- wmo/core/__init__.py +21 -0
- wmo/core/parsing.py +281 -0
- wmo/core/render.py +271 -0
- wmo/core/text.py +40 -0
- wmo/core/types.py +116 -0
- wmo/distill/__init__.py +14 -0
- wmo/distill/agents.py +140 -0
- wmo/distill/config.py +1006 -0
- wmo/distill/cost.py +437 -0
- wmo/distill/data.py +921 -0
- wmo/distill/deadlines.py +254 -0
- wmo/distill/fake_tinker.py +734 -0
- wmo/distill/gate.py +122 -0
- wmo/distill/loop.py +3499 -0
- wmo/distill/renderers.py +399 -0
- wmo/distill/rendering.py +620 -0
- wmo/distill/rollouts.py +726 -0
- wmo/distill/samples.py +195 -0
- wmo/distill/store.py +829 -0
- wmo/distill/teacher.py +714 -0
- wmo/distill/tokens.py +535 -0
- wmo/distill/tracking.py +552 -0
- wmo/distill/tripwire.py +411 -0
- wmo/distill/xtoken/byte_offsets.py +152 -0
- wmo/distill/xtoken/chunks.py +457 -0
- wmo/distill/xtoken/prompt_logprobs.py +475 -0
- wmo/distill/xtoken/teacher_render.py +346 -0
- wmo/engine/__init__.py +28 -0
- wmo/engine/autoconfig.py +367 -0
- wmo/engine/build.py +346 -0
- wmo/engine/demo.py +77 -0
- wmo/engine/eval_suites.py +245 -0
- wmo/engine/grounding.py +491 -0
- wmo/engine/knowledge.py +291 -0
- wmo/engine/loader.py +36 -0
- wmo/engine/play.py +92 -0
- wmo/engine/prompts.py +99 -0
- wmo/engine/replay.py +443 -0
- wmo/engine/reporting.py +58 -0
- wmo/engine/workspace.py +468 -0
- wmo/engine/world_model.py +568 -0
- wmo/env/__init__.py +22 -0
- wmo/env/base.py +121 -0
- wmo/env/closed_loop.py +229 -0
- wmo/env/episode.py +107 -0
- wmo/env/llm_agent.py +93 -0
- wmo/env/scenarios.py +73 -0
- wmo/evals/__init__.py +52 -0
- wmo/evals/agreement.py +110 -0
- wmo/evals/base.py +45 -0
- wmo/evals/closed_loop.py +480 -0
- wmo/evals/failover.py +96 -0
- wmo/evals/gold.py +127 -0
- wmo/evals/grid.py +394 -0
- wmo/evals/grid_plot.py +205 -0
- wmo/evals/harbor/__init__.py +27 -0
- wmo/evals/harbor/agent.py +573 -0
- wmo/evals/harbor/ctrf.py +171 -0
- wmo/evals/harbor/e2b_environment.py +587 -0
- wmo/evals/harbor/e2b_template_policy.py +144 -0
- wmo/evals/harbor/scorer.py +875 -0
- wmo/evals/harbor/tasks.py +140 -0
- wmo/evals/open_loop.py +194 -0
- wmo/evals/tasks.py +53 -0
- wmo/harness/__init__.py +51 -0
- wmo/harness/code_runtime.py +288 -0
- wmo/harness/create.py +1191 -0
- wmo/harness/delta.py +220 -0
- wmo/harness/doc.py +556 -0
- wmo/harness/e2b_ledger.py +342 -0
- wmo/harness/e2b_reap.py +476 -0
- wmo/harness/e2b_sandbox.py +350 -0
- wmo/harness/environment.py +35 -0
- wmo/harness/live_session.py +543 -0
- wmo/harness/mutate.py +343 -0
- wmo/harness/pi_e2b.py +1710 -0
- wmo/harness/pi_entry/entry.ts +268 -0
- wmo/harness/pi_entry/runner_frames.ts +92 -0
- wmo/harness/pi_entry/runner_live.ts +587 -0
- wmo/harness/pi_entry/runner_service.ts +270 -0
- wmo/harness/pi_entry/runner_stdio.ts +374 -0
- wmo/harness/pi_entry/runner_termination.ts +142 -0
- wmo/harness/pi_local.py +262 -0
- wmo/harness/pi_runtime.py +495 -0
- wmo/harness/pi_vendor.py +65 -0
- wmo/harness/population.py +509 -0
- wmo/harness/project_proposer.py +569 -0
- wmo/harness/proposer.py +977 -0
- wmo/harness/runner_link.py +619 -0
- wmo/harness/runtime.py +389 -0
- wmo/harness/scoring.py +247 -0
- wmo/harness/skills.py +116 -0
- wmo/harness/source_tree.py +319 -0
- wmo/harness/store.py +176 -0
- wmo/harness/tools.py +105 -0
- wmo/harness/vendor/manifest.sha256 +58 -0
- wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
- wmo/harness/vendor/pi-agent/LICENSE +21 -0
- wmo/harness/vendor/pi-agent/README.md +488 -0
- wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
- wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
- wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
- wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
- wmo/harness/vendor/pi-agent/docs/models.md +966 -0
- wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
- wmo/harness/vendor/pi-agent/package.json +60 -0
- wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
- wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
- wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
- wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
- wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
- wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
- wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
- wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
- wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
- wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
- wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
- wmo/harness/vendor/pi-agent/src/index.ts +44 -0
- wmo/harness/vendor/pi-agent/src/node.ts +2 -0
- wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
- wmo/harness/vendor/pi-agent/src/types.ts +428 -0
- wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
- wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
- wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
- wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
- wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
- wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
- wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
- wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
- wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
- wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
- wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
- wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
- wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
- wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
- wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
- wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
- wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
- wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
- wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
- wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
- wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
- wmo/harness/vendor/vendor_pi.sh +59 -0
- wmo/harness/workspace_patch.py +270 -0
- wmo/ingest/__init__.py +47 -0
- wmo/ingest/adapter.py +72 -0
- wmo/ingest/base.py +114 -0
- wmo/ingest/braintrust.py +339 -0
- wmo/ingest/detect.py +126 -0
- wmo/ingest/langfuse.py +291 -0
- wmo/ingest/langsmith.py +444 -0
- wmo/ingest/mastra.py +330 -0
- wmo/ingest/messages.py +170 -0
- wmo/ingest/normalize.py +679 -0
- wmo/ingest/otel_genai.py +69 -0
- wmo/ingest/otel_writer.py +100 -0
- wmo/ingest/phoenix.py +150 -0
- wmo/ingest/postgres.py +246 -0
- wmo/ingest/posthog.py +320 -0
- wmo/ingest/quality.py +28 -0
- wmo/ingest/stream.py +209 -0
- wmo/ingest/testdata/sample_otlp.json +60 -0
- wmo/ingest/testdata/sample_spans.jsonl +3 -0
- wmo/optimize/__init__.py +25 -0
- wmo/optimize/base.py +143 -0
- wmo/optimize/gepa.py +806 -0
- wmo/optimize/judge.py +262 -0
- wmo/optimize/judge_quality.py +359 -0
- wmo/optimize/knn.py +468 -0
- wmo/optimize/numeric.py +152 -0
- wmo/optimize/outcomes.py +103 -0
- wmo/optimize/policy.py +669 -0
- wmo/optimize/report.py +231 -0
- wmo/optimize/reward.py +129 -0
- wmo/optimize/routing.py +373 -0
- wmo/platform/__init__.py +6 -0
- wmo/platform/auth.py +115 -0
- wmo/platform/client.py +551 -0
- wmo/platform/credentials.py +126 -0
- wmo/platform/transfer.py +158 -0
- wmo/providers/__init__.py +40 -0
- wmo/providers/_bedrock_chat.py +155 -0
- wmo/providers/_openai_common.py +182 -0
- wmo/providers/_responses_common.py +472 -0
- wmo/providers/anthropic.py +134 -0
- wmo/providers/azure_openai.py +296 -0
- wmo/providers/base.py +300 -0
- wmo/providers/bedrock.py +312 -0
- wmo/providers/models.py +205 -0
- wmo/providers/openai.py +143 -0
- wmo/providers/openai_responses.py +240 -0
- wmo/providers/pool.py +170 -0
- wmo/providers/registry.py +73 -0
- wmo/providers/retry.py +151 -0
- wmo/providers/tinker.py +936 -0
- wmo/providers/waterfall.py +336 -0
- wmo/research/__init__.py +81 -0
- wmo/research/ablation.py +133 -0
- wmo/research/concurrency_plot.py +523 -0
- wmo/research/concurrency_run.py +240 -0
- wmo/research/concurrency_scaling.py +270 -0
- wmo/research/gepa_scaling.py +274 -0
- wmo/research/pipeline.py +198 -0
- wmo/research/scaling_split.py +82 -0
- wmo/research/scenario_fidelity.py +198 -0
- wmo/research/scenario_recovery.py +92 -0
- wmo/research/seed_stability.py +90 -0
- wmo/research/trace_scaling.py +348 -0
- wmo/retrieval/__init__.py +6 -0
- wmo/retrieval/embedders.py +105 -0
- wmo/retrieval/leakfree.py +52 -0
- wmo/retrieval/retriever.py +173 -0
- wmo/scenarios/__init__.py +58 -0
- wmo/scenarios/builder.py +152 -0
- wmo/scenarios/mining/__init__.py +27 -0
- wmo/scenarios/mining/clustering.py +171 -0
- wmo/scenarios/mining/facets.py +226 -0
- wmo/scenarios/mining/selection.py +220 -0
- wmo/scenarios/synthesis/__init__.py +6 -0
- wmo/scenarios/synthesis/scenario_set.py +63 -0
- wmo/scenarios/synthesis/synthesizer.py +85 -0
- wmo/scenarios/verification/__init__.py +17 -0
- wmo/scenarios/verification/judge.py +97 -0
- wmo/scenarios/verification/verify.py +135 -0
- wmo/serving/__init__.py +5 -0
- wmo/serving/builds.py +451 -0
- wmo/serving/chat.py +878 -0
- wmo/serving/endpoint_config.py +64 -0
- wmo/serving/savings.py +250 -0
- wmo/serving/server.py +553 -0
- wmo/serving/traces_source.py +206 -0
- wmo/telemetry.py +213 -0
- wmo/tracking/__init__.py +36 -0
- wmo/tracking/clock.py +24 -0
- wmo/tracking/metered.py +125 -0
- wmo/tracking/pricing.py +99 -0
- wmo/tracking/store.py +31 -0
- wmo/tracking/tracker.py +149 -0
- world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
- world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
- world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
- world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,350 @@
|
|
|
1
|
+
"""E2B sandbox plumbing for the e2b harness backend: protocol slice, creation, retries.
|
|
2
|
+
|
|
3
|
+
An E2B microVM is where a `pi-node` harness *process* executes under `backend="e2b"` — the
|
|
4
|
+
environment its tool calls hit stays whatever `AgentEnvironment` the eval binds (normally the
|
|
5
|
+
world-model simulation). This module owns only the sandbox mechanics: the exact protocol slice of
|
|
6
|
+
`e2b.Sandbox` the harness uses (so tests substitute fakes), the lazy-SDK default factory, and
|
|
7
|
+
capacity-shaped creation retries with fixed (1, 3, 9) s delays — the
|
|
8
|
+
`wmo.providers.retry.RetryingProvider` precedent, no RNG in scoring paths. The e2b SDK stays an
|
|
9
|
+
optional extra (`uv sync --extra e2b`).
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import asyncio
|
|
15
|
+
import os
|
|
16
|
+
import time
|
|
17
|
+
from collections.abc import Callable, Iterator, Sequence
|
|
18
|
+
from threading import Lock
|
|
19
|
+
from typing import Protocol, cast, runtime_checkable
|
|
20
|
+
|
|
21
|
+
from pydantic import BaseModel
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class SandboxUsage(BaseModel):
|
|
25
|
+
"""E2B spend metrics for a pool: how many sandboxes ran, and their total lifetime seconds.
|
|
26
|
+
|
|
27
|
+
Seconds are wall-clock sandbox lifetimes (create -> kill; live sandboxes count up to now),
|
|
28
|
+
the unit E2B bills on. Pricing is the caller's concern (deployment-specific instance rates);
|
|
29
|
+
this is the raw meter.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
count: int = 0
|
|
33
|
+
seconds: float = 0.0
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
E2B_API_KEY_ENV = "E2B_API_KEY"
|
|
37
|
+
E2B_TEMPLATE_ENV = "WMO_E2B_TEMPLATE"
|
|
38
|
+
|
|
39
|
+
# Sandbox lifetime. The sandbox only hosts the harness process (tool calls are answered by the
|
|
40
|
+
# environment host-side), so the bound is episode wall-time, not command time.
|
|
41
|
+
DEFAULT_SANDBOX_TIMEOUT_S = 900.0
|
|
42
|
+
|
|
43
|
+
# Fixed delays before each retry of sandbox creation (RetryingProvider precedent: 1s, 3s, 9s).
|
|
44
|
+
_CREATE_DELAYS = (1.0, 3.0, 9.0)
|
|
45
|
+
# Teardown is normally one cheap request. Two short deterministic retries cover a stale HTTP/2
|
|
46
|
+
# connection without adding latency to the success path or hiding a sandbox whose release cannot
|
|
47
|
+
# be proved.
|
|
48
|
+
_KILL_DELAYS = (0.1, 0.5)
|
|
49
|
+
_KILL_REQUEST_TIMEOUT_S = 5.0
|
|
50
|
+
|
|
51
|
+
# E2B publishes a 5/sec sandbox-create rate limit per account; admit at 4/sec so every consumer
|
|
52
|
+
# in this process (the pi-node worker pool AND harbor E2B task environments) shares headroom.
|
|
53
|
+
E2B_CREATES_PER_SECOND = 4
|
|
54
|
+
_E2B_CREATE_INTERVAL_NS = 1_000_000_000 // E2B_CREATES_PER_SECOND
|
|
55
|
+
_E2B_CREATE_MAX_WAIT_S = 60.0
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class E2BCreateRateLimitError(RuntimeError):
|
|
59
|
+
"""A sandbox create could not be admitted within the bounded queue wait."""
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class E2BCreateRateGate:
|
|
63
|
+
"""Serialize one process's E2B sandbox creates at a fixed monotonic rate.
|
|
64
|
+
|
|
65
|
+
One lock plus a next-admission timestamp: each acquire sleeps until its scheduled slot and
|
|
66
|
+
schedules the next one `interval_ns` later. Waiters queue on the lock, so the admission
|
|
67
|
+
schedule stays monotonic under concurrency. A caller whose computed wait exceeds
|
|
68
|
+
`max_wait_s` fails fast instead of silently queueing behind a create storm.
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
def __init__(
|
|
72
|
+
self,
|
|
73
|
+
*,
|
|
74
|
+
interval_ns: int = _E2B_CREATE_INTERVAL_NS,
|
|
75
|
+
max_wait_s: float = _E2B_CREATE_MAX_WAIT_S,
|
|
76
|
+
monotonic_ns: Callable[[], int] = time.monotonic_ns,
|
|
77
|
+
sleep: Callable[[float], None] = time.sleep,
|
|
78
|
+
) -> None:
|
|
79
|
+
if interval_ns < 1 or max_wait_s <= 0:
|
|
80
|
+
raise ValueError("E2B create-rate gate values must be positive")
|
|
81
|
+
self._interval_ns = interval_ns
|
|
82
|
+
self._max_wait_s = max_wait_s
|
|
83
|
+
self._monotonic_ns = monotonic_ns
|
|
84
|
+
self._sleep = sleep
|
|
85
|
+
self._lock = Lock()
|
|
86
|
+
self._next_admission_ns: int | None = None
|
|
87
|
+
|
|
88
|
+
def acquire(self) -> None:
|
|
89
|
+
"""Wait for one create slot or fail before the bounded queue horizon."""
|
|
90
|
+
started_ns = self._monotonic_ns()
|
|
91
|
+
with self._lock:
|
|
92
|
+
now_ns = self._monotonic_ns()
|
|
93
|
+
target_ns = max(now_ns, self._next_admission_ns or now_ns)
|
|
94
|
+
if target_ns - started_ns > round(self._max_wait_s * 1_000_000_000):
|
|
95
|
+
raise E2BCreateRateLimitError(
|
|
96
|
+
f"E2B sandbox create could not be admitted within {self._max_wait_s:.3f} "
|
|
97
|
+
"seconds; reduce concurrent sandbox creation or stagger score waves"
|
|
98
|
+
)
|
|
99
|
+
while now_ns < target_ns:
|
|
100
|
+
self._sleep((target_ns - now_ns) / 1_000_000_000)
|
|
101
|
+
now_ns = self._monotonic_ns()
|
|
102
|
+
self._next_admission_ns = max(now_ns, target_ns) + self._interval_ns
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
_E2B_CREATE_RATE_GATE = E2BCreateRateGate()
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def acquire_e2b_create_slot() -> None:
|
|
109
|
+
"""Acquire one slot from the process-wide E2B sandbox-create gate."""
|
|
110
|
+
_E2B_CREATE_RATE_GATE.acquire()
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
async def acquire_e2b_create_slot_async() -> None:
|
|
114
|
+
"""Acquire the same process-wide slot without blocking the event loop."""
|
|
115
|
+
await asyncio.to_thread(acquire_e2b_create_slot)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def resolve_e2b_template(template: str | None) -> str | None:
|
|
119
|
+
"""Resolve the template once; an explicit empty string disables environment fallback."""
|
|
120
|
+
if template is None:
|
|
121
|
+
return os.environ.get(E2B_TEMPLATE_ENV) or None
|
|
122
|
+
return template or None
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
class SandboxCleanupError(RuntimeError):
|
|
126
|
+
"""An E2B sandbox may still be live after bounded teardown retries."""
|
|
127
|
+
|
|
128
|
+
def __init__(
|
|
129
|
+
self,
|
|
130
|
+
message: str,
|
|
131
|
+
*,
|
|
132
|
+
resource: str = "e2b_sandbox",
|
|
133
|
+
sandbox_usage: SandboxUsage | None = None,
|
|
134
|
+
) -> None:
|
|
135
|
+
super().__init__(message)
|
|
136
|
+
self.resource = resource
|
|
137
|
+
self.sandbox_usage = sandbox_usage
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
@runtime_checkable
|
|
141
|
+
class CommandOutput(Protocol):
|
|
142
|
+
"""The result slice of a finished sandbox command (e2b's `CommandResult` shape).
|
|
143
|
+
|
|
144
|
+
`runtime_checkable` because e2b's `CommandExitException` *is* a `CommandResult` (non-zero
|
|
145
|
+
exits raise instead of returning) — an isinstance check against this protocol recognizes it
|
|
146
|
+
without importing the SDK.
|
|
147
|
+
"""
|
|
148
|
+
|
|
149
|
+
stdout: str
|
|
150
|
+
stderr: str
|
|
151
|
+
exit_code: int
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
class CommandHandle(Protocol):
|
|
155
|
+
"""A background sandbox command (e2b's handle): stdin by pid, iteration yields stream events.
|
|
156
|
+
|
|
157
|
+
Iteration events are `(stdout, stderr, pty)` chunks; `E2BPiRuntime` drives the RunnerLink
|
|
158
|
+
frame stream over one.
|
|
159
|
+
"""
|
|
160
|
+
|
|
161
|
+
@property
|
|
162
|
+
def pid(self) -> int: ...
|
|
163
|
+
|
|
164
|
+
def __iter__(self) -> Iterator[tuple[str | None, str | None, str | None]]: ...
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
class SandboxCommands(Protocol):
|
|
168
|
+
"""The `sandbox.commands` slice: run/connect commands and inject stdin."""
|
|
169
|
+
|
|
170
|
+
def run(
|
|
171
|
+
self,
|
|
172
|
+
cmd: str,
|
|
173
|
+
background: bool | None = None,
|
|
174
|
+
*,
|
|
175
|
+
envs: dict[str, str] | None = None,
|
|
176
|
+
stdin: bool | None = None,
|
|
177
|
+
timeout: float | None = None,
|
|
178
|
+
) -> CommandOutput | CommandHandle: ...
|
|
179
|
+
|
|
180
|
+
def connect(
|
|
181
|
+
self,
|
|
182
|
+
pid: int,
|
|
183
|
+
*,
|
|
184
|
+
timeout: float | None = None,
|
|
185
|
+
) -> CommandHandle: ...
|
|
186
|
+
|
|
187
|
+
def send_stdin(
|
|
188
|
+
self,
|
|
189
|
+
pid: int,
|
|
190
|
+
data: str,
|
|
191
|
+
request_timeout: float | None = None,
|
|
192
|
+
) -> object: ...
|
|
193
|
+
|
|
194
|
+
def list(self, request_timeout: float | None = None) -> Sequence[SandboxProcess]: ...
|
|
195
|
+
|
|
196
|
+
def kill(self, pid: int, request_timeout: float | None = None) -> object: ...
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
class SandboxProcess(Protocol):
|
|
200
|
+
"""The running-process field used to classify a durable runner stream EOF."""
|
|
201
|
+
|
|
202
|
+
@property
|
|
203
|
+
def pid(self) -> int: ...
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
class SandboxFiles(Protocol):
|
|
207
|
+
"""The `sandbox.files` slice: whole-file read and write."""
|
|
208
|
+
|
|
209
|
+
def write(self, path: str, data: str) -> object: ...
|
|
210
|
+
|
|
211
|
+
def read(
|
|
212
|
+
self,
|
|
213
|
+
path: str,
|
|
214
|
+
*,
|
|
215
|
+
request_timeout: float | None = None,
|
|
216
|
+
gzip: bool = False,
|
|
217
|
+
) -> str: ...
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
@runtime_checkable
|
|
221
|
+
class SandboxHandle(Protocol):
|
|
222
|
+
"""The exact slice of `e2b.Sandbox` the harness uses, so tests substitute fakes."""
|
|
223
|
+
|
|
224
|
+
@property
|
|
225
|
+
def commands(self) -> SandboxCommands: ...
|
|
226
|
+
|
|
227
|
+
@property
|
|
228
|
+
def files(self) -> SandboxFiles: ...
|
|
229
|
+
|
|
230
|
+
def set_timeout(self, timeout: int) -> None: ...
|
|
231
|
+
|
|
232
|
+
def kill(self, request_timeout: float | None = None) -> object: ...
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
# Opens one sandbox. The default factory calls the real SDK; tests inject fakes.
|
|
236
|
+
SandboxFactory = Callable[[], SandboxHandle]
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def default_sandbox_factory(
|
|
240
|
+
*,
|
|
241
|
+
api_key: str | None = None,
|
|
242
|
+
template: str | None = None,
|
|
243
|
+
timeout: float = DEFAULT_SANDBOX_TIMEOUT_S,
|
|
244
|
+
metadata: dict[str, str] | None = None,
|
|
245
|
+
) -> SandboxFactory:
|
|
246
|
+
"""A factory creating real E2B sandboxes (lazy SDK import; key from arg or $E2B_API_KEY).
|
|
247
|
+
|
|
248
|
+
`metadata` tags the sandbox at create time (e.g. `{"session_id": …}`) so an out-of-band sweep
|
|
249
|
+
(`Sandbox.list`) can find and reap an orphaned sandbox whose owning process died — the live
|
|
250
|
+
session driver relies on this for cost-leak reconciliation.
|
|
251
|
+
"""
|
|
252
|
+
# Snapshot once at construction so a factory built under one $WMO_E2B_TEMPLATE never drifts
|
|
253
|
+
# to a different template mid-search when the environment changes underneath it.
|
|
254
|
+
chosen_template = resolve_e2b_template(template)
|
|
255
|
+
|
|
256
|
+
def make() -> SandboxHandle:
|
|
257
|
+
try:
|
|
258
|
+
from e2b import Sandbox
|
|
259
|
+
except ImportError as exc: # pragma: no cover - exercised only without the extra
|
|
260
|
+
raise ImportError(
|
|
261
|
+
"the e2b SDK is not installed; run `uv sync --extra e2b` to use the "
|
|
262
|
+
"e2b harness backend"
|
|
263
|
+
) from exc
|
|
264
|
+
key = api_key or os.environ.get(E2B_API_KEY_ENV)
|
|
265
|
+
if not key:
|
|
266
|
+
raise RuntimeError(f"set ${E2B_API_KEY_ENV} to run the harness in E2B sandboxes")
|
|
267
|
+
# Every create attempt (including capacity retries) re-enters the shared process gate:
|
|
268
|
+
# a retry storm must not burst past the account-wide create rate.
|
|
269
|
+
acquire_e2b_create_slot()
|
|
270
|
+
if metadata:
|
|
271
|
+
sandbox = Sandbox.create(
|
|
272
|
+
template=chosen_template, timeout=int(timeout), api_key=key, metadata=metadata
|
|
273
|
+
)
|
|
274
|
+
else:
|
|
275
|
+
sandbox = Sandbox.create(template=chosen_template, timeout=int(timeout), api_key=key)
|
|
276
|
+
# The SDK object satisfies the protocol slice structurally; cast rather than pin the
|
|
277
|
+
# SDK's full (much wider) signatures into the protocol.
|
|
278
|
+
return cast("SandboxHandle", sandbox)
|
|
279
|
+
|
|
280
|
+
return make
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def create_sandbox(factory: SandboxFactory) -> SandboxHandle:
|
|
284
|
+
"""Open one sandbox via `factory`, retrying capacity errors with fixed (1, 3, 9) s delays."""
|
|
285
|
+
for delay in _CREATE_DELAYS:
|
|
286
|
+
try:
|
|
287
|
+
return factory()
|
|
288
|
+
except Exception as exc: # noqa: BLE001 - classified below; non-capacity re-raises
|
|
289
|
+
if not _is_retryable_create_error(exc):
|
|
290
|
+
raise
|
|
291
|
+
time.sleep(delay)
|
|
292
|
+
return factory() # final attempt: let any error propagate
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def kill_sandbox(sandbox: SandboxHandle) -> None:
|
|
296
|
+
"""Kill one sandbox with bounded retries, failing closed when release is unproven.
|
|
297
|
+
|
|
298
|
+
A successful call, a falsey SDK result, or an explicit already-gone response all mean there
|
|
299
|
+
is no live resource left to meter. Other exceptions are retried twice; exhausting that bound
|
|
300
|
+
raises :class:`SandboxCleanupError` so callers cannot report clean cancellation while the
|
|
301
|
+
sandbox may still be billable.
|
|
302
|
+
"""
|
|
303
|
+
for delay in _KILL_DELAYS:
|
|
304
|
+
try:
|
|
305
|
+
sandbox.kill(request_timeout=_KILL_REQUEST_TIMEOUT_S)
|
|
306
|
+
return
|
|
307
|
+
except Exception as error: # noqa: BLE001 - E2B SDK errors are optional/import-free here
|
|
308
|
+
if _is_already_gone_error(error):
|
|
309
|
+
return
|
|
310
|
+
time.sleep(delay)
|
|
311
|
+
try:
|
|
312
|
+
sandbox.kill(request_timeout=_KILL_REQUEST_TIMEOUT_S)
|
|
313
|
+
except Exception as error: # noqa: BLE001 - promote the bounded cleanup failure uniformly
|
|
314
|
+
if _is_already_gone_error(error):
|
|
315
|
+
return
|
|
316
|
+
sandbox_id = getattr(sandbox, "sandbox_id", None) or getattr(sandbox, "id", None)
|
|
317
|
+
identity = f" {sandbox_id!r}" if sandbox_id is not None else ""
|
|
318
|
+
raise SandboxCleanupError(
|
|
319
|
+
f"E2B sandbox{identity} cleanup failed after {len(_KILL_DELAYS) + 1} attempts: {error}"
|
|
320
|
+
) from error
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def _is_retryable_create_error(exc: Exception) -> bool:
|
|
324
|
+
"""True for capacity-shaped creation failures (rate limit / no capacity / 5xx).
|
|
325
|
+
|
|
326
|
+
Matched by exception name and message so fakes need no SDK import; anything else (auth,
|
|
327
|
+
bad template, missing key) fails immediately — retrying those only hides real bugs.
|
|
328
|
+
"""
|
|
329
|
+
if type(exc).__name__ == "RateLimitException": # e2b's 429
|
|
330
|
+
return True
|
|
331
|
+
text = str(exc).lower()
|
|
332
|
+
if "rate limit" in text or "capacity" in text or "too many requests" in text:
|
|
333
|
+
return True
|
|
334
|
+
return any(code in text for code in ("429", "500", "502", "503", "504"))
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def _is_already_gone_error(exc: Exception) -> bool:
|
|
338
|
+
"""Whether a failed kill explicitly proves the sandbox no longer exists."""
|
|
339
|
+
text = str(exc).lower()
|
|
340
|
+
return any(
|
|
341
|
+
marker in text
|
|
342
|
+
for marker in (
|
|
343
|
+
"sandbox not found",
|
|
344
|
+
"sandbox is not found",
|
|
345
|
+
"sandbox already killed",
|
|
346
|
+
"sandbox has been killed",
|
|
347
|
+
"sandbox already closed",
|
|
348
|
+
"sandbox has expired",
|
|
349
|
+
)
|
|
350
|
+
)
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""The environment seam: the agent loop talks to an interface, not to any backend directly.
|
|
2
|
+
|
|
3
|
+
The `AgentEnvironment` protocol is the substitution point: closed-loop eval binds it to the world
|
|
4
|
+
model (`wmo.evals.closed_loop.WorldModelEnvironment` — every tool call answered by
|
|
5
|
+
`WorldModel.step`), and a real execution backend (a managed sandbox) implements the same two
|
|
6
|
+
methods, so the *same* agent loop and scoring can run against reality when one is available. That
|
|
7
|
+
symmetry is what makes a simulated report comparable to a real one (`wmo.evals.agreement`).
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from typing import Protocol, runtime_checkable
|
|
13
|
+
|
|
14
|
+
from wmo.core.types import Action, ActionKind, Observation
|
|
15
|
+
|
|
16
|
+
# The tool names the environment answers (everything except the runtime-handled `submit`).
|
|
17
|
+
ENV_TOOLS = frozenset({"bash", "read_file", "write_file"})
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@runtime_checkable
|
|
21
|
+
class AgentEnvironment(Protocol):
|
|
22
|
+
"""Executes an agent Action and returns what the environment observed."""
|
|
23
|
+
|
|
24
|
+
def execute(self, action: Action) -> Observation:
|
|
25
|
+
"""Run one action; return the resulting observation."""
|
|
26
|
+
...
|
|
27
|
+
|
|
28
|
+
def close(self) -> None:
|
|
29
|
+
"""Release any underlying resources (end the session)."""
|
|
30
|
+
...
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def is_env_action(action: Action) -> bool:
|
|
34
|
+
"""True when the action is one the environment answers (a tool call to an env tool)."""
|
|
35
|
+
return action.kind == ActionKind.TOOL_CALL and action.name in ENV_TOOLS
|