world-model-optimizer 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_waterfall/LICENSE +21 -0
- llm_waterfall/__init__.py +53 -0
- llm_waterfall/adapters/__init__.py +36 -0
- llm_waterfall/adapters/anthropic.py +105 -0
- llm_waterfall/adapters/aws_mantle.py +47 -0
- llm_waterfall/adapters/azure_openai.py +71 -0
- llm_waterfall/adapters/base.py +51 -0
- llm_waterfall/adapters/bedrock.py +309 -0
- llm_waterfall/adapters/openai.py +130 -0
- llm_waterfall/classify.py +184 -0
- llm_waterfall/pricing.py +110 -0
- llm_waterfall/py.typed +0 -0
- llm_waterfall/types.py +295 -0
- llm_waterfall/waterfall.py +255 -0
- wmo/__init__.py +38 -0
- wmo/agents/__init__.py +7 -0
- wmo/agents/default.py +29 -0
- wmo/agents/meta.py +55 -0
- wmo/agents/optimizer.py +55 -0
- wmo/agents/project.py +928 -0
- wmo/cli/__init__.py +5 -0
- wmo/cli/agent_session.py +1123 -0
- wmo/cli/app.py +2489 -0
- wmo/cli/e2b_cmds.py +212 -0
- wmo/cli/eval_closed_loop.py +207 -0
- wmo/cli/harness_app.py +1147 -0
- wmo/cli/harness_distill.py +659 -0
- wmo/cli/hosted_session.py +880 -0
- wmo/cli/ingest_cmd.py +165 -0
- wmo/cli/model_roles.py +82 -0
- wmo/cli/platform_cmds.py +372 -0
- wmo/cli/route_app.py +274 -0
- wmo/cli/session_state.py +243 -0
- wmo/cli/ui.py +1107 -0
- wmo/cli/workspace_sync.py +504 -0
- wmo/config/__init__.py +60 -0
- wmo/config/card.py +129 -0
- wmo/config/config.py +367 -0
- wmo/config/dotenv.py +67 -0
- wmo/config/settings.py +128 -0
- wmo/config/store.py +177 -0
- wmo/conftest.py +19 -0
- wmo/connect/__init__.py +88 -0
- wmo/connect/apps.py +78 -0
- wmo/connect/brave.py +284 -0
- wmo/connect/connector.py +79 -0
- wmo/connect/credentials.py +164 -0
- wmo/connect/github.py +321 -0
- wmo/connect/google.py +627 -0
- wmo/connect/notion.py +790 -0
- wmo/connect/oauth.py +461 -0
- wmo/connect/slack.py +555 -0
- wmo/connect/store.py +199 -0
- wmo/connect/types.py +156 -0
- wmo/core/__init__.py +21 -0
- wmo/core/parsing.py +281 -0
- wmo/core/render.py +271 -0
- wmo/core/text.py +40 -0
- wmo/core/types.py +116 -0
- wmo/distill/__init__.py +14 -0
- wmo/distill/agents.py +140 -0
- wmo/distill/config.py +1006 -0
- wmo/distill/cost.py +437 -0
- wmo/distill/data.py +921 -0
- wmo/distill/deadlines.py +254 -0
- wmo/distill/fake_tinker.py +734 -0
- wmo/distill/gate.py +122 -0
- wmo/distill/loop.py +3499 -0
- wmo/distill/renderers.py +399 -0
- wmo/distill/rendering.py +620 -0
- wmo/distill/rollouts.py +726 -0
- wmo/distill/samples.py +195 -0
- wmo/distill/store.py +829 -0
- wmo/distill/teacher.py +714 -0
- wmo/distill/tokens.py +535 -0
- wmo/distill/tracking.py +552 -0
- wmo/distill/tripwire.py +411 -0
- wmo/distill/xtoken/byte_offsets.py +152 -0
- wmo/distill/xtoken/chunks.py +457 -0
- wmo/distill/xtoken/prompt_logprobs.py +475 -0
- wmo/distill/xtoken/teacher_render.py +346 -0
- wmo/engine/__init__.py +28 -0
- wmo/engine/autoconfig.py +367 -0
- wmo/engine/build.py +346 -0
- wmo/engine/demo.py +77 -0
- wmo/engine/eval_suites.py +245 -0
- wmo/engine/grounding.py +491 -0
- wmo/engine/knowledge.py +291 -0
- wmo/engine/loader.py +36 -0
- wmo/engine/play.py +92 -0
- wmo/engine/prompts.py +99 -0
- wmo/engine/replay.py +443 -0
- wmo/engine/reporting.py +58 -0
- wmo/engine/workspace.py +468 -0
- wmo/engine/world_model.py +568 -0
- wmo/env/__init__.py +22 -0
- wmo/env/base.py +121 -0
- wmo/env/closed_loop.py +229 -0
- wmo/env/episode.py +107 -0
- wmo/env/llm_agent.py +93 -0
- wmo/env/scenarios.py +73 -0
- wmo/evals/__init__.py +52 -0
- wmo/evals/agreement.py +110 -0
- wmo/evals/base.py +45 -0
- wmo/evals/closed_loop.py +480 -0
- wmo/evals/failover.py +96 -0
- wmo/evals/gold.py +127 -0
- wmo/evals/grid.py +394 -0
- wmo/evals/grid_plot.py +205 -0
- wmo/evals/harbor/__init__.py +27 -0
- wmo/evals/harbor/agent.py +573 -0
- wmo/evals/harbor/ctrf.py +171 -0
- wmo/evals/harbor/e2b_environment.py +587 -0
- wmo/evals/harbor/e2b_template_policy.py +144 -0
- wmo/evals/harbor/scorer.py +875 -0
- wmo/evals/harbor/tasks.py +140 -0
- wmo/evals/open_loop.py +194 -0
- wmo/evals/tasks.py +53 -0
- wmo/harness/__init__.py +51 -0
- wmo/harness/code_runtime.py +288 -0
- wmo/harness/create.py +1191 -0
- wmo/harness/delta.py +220 -0
- wmo/harness/doc.py +556 -0
- wmo/harness/e2b_ledger.py +342 -0
- wmo/harness/e2b_reap.py +476 -0
- wmo/harness/e2b_sandbox.py +350 -0
- wmo/harness/environment.py +35 -0
- wmo/harness/live_session.py +543 -0
- wmo/harness/mutate.py +343 -0
- wmo/harness/pi_e2b.py +1710 -0
- wmo/harness/pi_entry/entry.ts +268 -0
- wmo/harness/pi_entry/runner_frames.ts +92 -0
- wmo/harness/pi_entry/runner_live.ts +587 -0
- wmo/harness/pi_entry/runner_service.ts +270 -0
- wmo/harness/pi_entry/runner_stdio.ts +374 -0
- wmo/harness/pi_entry/runner_termination.ts +142 -0
- wmo/harness/pi_local.py +262 -0
- wmo/harness/pi_runtime.py +495 -0
- wmo/harness/pi_vendor.py +65 -0
- wmo/harness/population.py +509 -0
- wmo/harness/project_proposer.py +569 -0
- wmo/harness/proposer.py +977 -0
- wmo/harness/runner_link.py +619 -0
- wmo/harness/runtime.py +389 -0
- wmo/harness/scoring.py +247 -0
- wmo/harness/skills.py +116 -0
- wmo/harness/source_tree.py +319 -0
- wmo/harness/store.py +176 -0
- wmo/harness/tools.py +105 -0
- wmo/harness/vendor/manifest.sha256 +58 -0
- wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
- wmo/harness/vendor/pi-agent/LICENSE +21 -0
- wmo/harness/vendor/pi-agent/README.md +488 -0
- wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
- wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
- wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
- wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
- wmo/harness/vendor/pi-agent/docs/models.md +966 -0
- wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
- wmo/harness/vendor/pi-agent/package.json +60 -0
- wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
- wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
- wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
- wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
- wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
- wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
- wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
- wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
- wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
- wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
- wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
- wmo/harness/vendor/pi-agent/src/index.ts +44 -0
- wmo/harness/vendor/pi-agent/src/node.ts +2 -0
- wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
- wmo/harness/vendor/pi-agent/src/types.ts +428 -0
- wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
- wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
- wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
- wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
- wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
- wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
- wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
- wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
- wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
- wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
- wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
- wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
- wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
- wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
- wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
- wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
- wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
- wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
- wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
- wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
- wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
- wmo/harness/vendor/vendor_pi.sh +59 -0
- wmo/harness/workspace_patch.py +270 -0
- wmo/ingest/__init__.py +47 -0
- wmo/ingest/adapter.py +72 -0
- wmo/ingest/base.py +114 -0
- wmo/ingest/braintrust.py +339 -0
- wmo/ingest/detect.py +126 -0
- wmo/ingest/langfuse.py +291 -0
- wmo/ingest/langsmith.py +444 -0
- wmo/ingest/mastra.py +330 -0
- wmo/ingest/messages.py +170 -0
- wmo/ingest/normalize.py +679 -0
- wmo/ingest/otel_genai.py +69 -0
- wmo/ingest/otel_writer.py +100 -0
- wmo/ingest/phoenix.py +150 -0
- wmo/ingest/postgres.py +246 -0
- wmo/ingest/posthog.py +320 -0
- wmo/ingest/quality.py +28 -0
- wmo/ingest/stream.py +209 -0
- wmo/ingest/testdata/sample_otlp.json +60 -0
- wmo/ingest/testdata/sample_spans.jsonl +3 -0
- wmo/optimize/__init__.py +25 -0
- wmo/optimize/base.py +143 -0
- wmo/optimize/gepa.py +806 -0
- wmo/optimize/judge.py +262 -0
- wmo/optimize/judge_quality.py +359 -0
- wmo/optimize/knn.py +468 -0
- wmo/optimize/numeric.py +152 -0
- wmo/optimize/outcomes.py +103 -0
- wmo/optimize/policy.py +669 -0
- wmo/optimize/report.py +231 -0
- wmo/optimize/reward.py +129 -0
- wmo/optimize/routing.py +373 -0
- wmo/platform/__init__.py +6 -0
- wmo/platform/auth.py +115 -0
- wmo/platform/client.py +551 -0
- wmo/platform/credentials.py +126 -0
- wmo/platform/transfer.py +158 -0
- wmo/providers/__init__.py +40 -0
- wmo/providers/_bedrock_chat.py +155 -0
- wmo/providers/_openai_common.py +182 -0
- wmo/providers/_responses_common.py +472 -0
- wmo/providers/anthropic.py +134 -0
- wmo/providers/azure_openai.py +296 -0
- wmo/providers/base.py +300 -0
- wmo/providers/bedrock.py +312 -0
- wmo/providers/models.py +205 -0
- wmo/providers/openai.py +143 -0
- wmo/providers/openai_responses.py +240 -0
- wmo/providers/pool.py +170 -0
- wmo/providers/registry.py +73 -0
- wmo/providers/retry.py +151 -0
- wmo/providers/tinker.py +936 -0
- wmo/providers/waterfall.py +336 -0
- wmo/research/__init__.py +81 -0
- wmo/research/ablation.py +133 -0
- wmo/research/concurrency_plot.py +523 -0
- wmo/research/concurrency_run.py +240 -0
- wmo/research/concurrency_scaling.py +270 -0
- wmo/research/gepa_scaling.py +274 -0
- wmo/research/pipeline.py +198 -0
- wmo/research/scaling_split.py +82 -0
- wmo/research/scenario_fidelity.py +198 -0
- wmo/research/scenario_recovery.py +92 -0
- wmo/research/seed_stability.py +90 -0
- wmo/research/trace_scaling.py +348 -0
- wmo/retrieval/__init__.py +6 -0
- wmo/retrieval/embedders.py +105 -0
- wmo/retrieval/leakfree.py +52 -0
- wmo/retrieval/retriever.py +173 -0
- wmo/scenarios/__init__.py +58 -0
- wmo/scenarios/builder.py +152 -0
- wmo/scenarios/mining/__init__.py +27 -0
- wmo/scenarios/mining/clustering.py +171 -0
- wmo/scenarios/mining/facets.py +226 -0
- wmo/scenarios/mining/selection.py +220 -0
- wmo/scenarios/synthesis/__init__.py +6 -0
- wmo/scenarios/synthesis/scenario_set.py +63 -0
- wmo/scenarios/synthesis/synthesizer.py +85 -0
- wmo/scenarios/verification/__init__.py +17 -0
- wmo/scenarios/verification/judge.py +97 -0
- wmo/scenarios/verification/verify.py +135 -0
- wmo/serving/__init__.py +5 -0
- wmo/serving/builds.py +451 -0
- wmo/serving/chat.py +878 -0
- wmo/serving/endpoint_config.py +64 -0
- wmo/serving/savings.py +250 -0
- wmo/serving/server.py +553 -0
- wmo/serving/traces_source.py +206 -0
- wmo/telemetry.py +213 -0
- wmo/tracking/__init__.py +36 -0
- wmo/tracking/clock.py +24 -0
- wmo/tracking/metered.py +125 -0
- wmo/tracking/pricing.py +99 -0
- wmo/tracking/store.py +31 -0
- wmo/tracking/tracker.py +149 -0
- world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
- world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
- world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
- world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,288 @@
|
|
|
1
|
+
"""`CodeRuntime`: the agent loop itself as a searchable harness surface.
|
|
2
|
+
|
|
3
|
+
Two live search campaigns showed the same wall: the meta-agent diagnoses failure mechanisms
|
|
4
|
+
correctly, but prompt- and skill-level edits cannot express the fixes that actually make a
|
|
5
|
+
harness good — loop structure, retries, context compaction, observation truncation, token
|
|
6
|
+
budgets. Those are *code*. A `code:runtime` surface holds a Python module defining
|
|
7
|
+
`run(kit) -> str`; the search edits that program.
|
|
8
|
+
|
|
9
|
+
The contract is the `RuntimeKit`, and it carries three guarantees the search relies on:
|
|
10
|
+
|
|
11
|
+
- **Budgeted.** `kit.complete` and `kit.execute` enforce hard caps on LLM calls and environment
|
|
12
|
+
actions. A runaway loop raises `BudgetExceeded` instead of wedging an eval; cost is bounded per
|
|
13
|
+
episode by construction, not by hope.
|
|
14
|
+
- **Kit-recorded.** Every environment action is appended to the transcript by the kit itself, so
|
|
15
|
+
the judge always scores ground truth: generated code cannot fake, omit, or reorder what it did.
|
|
16
|
+
- **Crash-isolated.** An exception inside `run` fails that episode (scored as a failure with the
|
|
17
|
+
partial transcript), never the evaluation loop around it.
|
|
18
|
+
|
|
19
|
+
The kit is an interface contract, not a security boundary: harness code runs in-process and is
|
|
20
|
+
trusted to the same degree as the rest of the search (it is reviewed via the delta archive, and
|
|
21
|
+
only ever exercised against the world model during search). Running searched code against real
|
|
22
|
+
environments is a deployment decision that belongs behind a sandbox.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
from collections.abc import Callable
|
|
28
|
+
from typing import cast
|
|
29
|
+
|
|
30
|
+
from pydantic import BaseModel, Field
|
|
31
|
+
|
|
32
|
+
from wmo.core.types import Action, ActionKind, EnvState, JsonObject, Observation, Step
|
|
33
|
+
from wmo.harness.environment import AgentEnvironment, is_env_action
|
|
34
|
+
from wmo.harness.runtime import RunResult, StopReason
|
|
35
|
+
from wmo.harness.skills import SkillLibrary
|
|
36
|
+
from wmo.harness.tools import ToolCall, ToolSpec, parse_tool_call, render_tools
|
|
37
|
+
from wmo.providers.base import Message, Provider
|
|
38
|
+
|
|
39
|
+
CODE_ENTRYPOINT = "run"
|
|
40
|
+
|
|
41
|
+
# Defaults sized so a reasonable loop never notices them: the fixed baseline loop uses at most
|
|
42
|
+
# `max_turns` (20) of each.
|
|
43
|
+
DEFAULT_MAX_LLM_CALLS = 40
|
|
44
|
+
DEFAULT_MAX_ENV_ACTIONS = 40
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class RunBudget(BaseModel):
|
|
48
|
+
"""Hard per-episode caps enforced by the kit."""
|
|
49
|
+
|
|
50
|
+
max_llm_calls: int = Field(default=DEFAULT_MAX_LLM_CALLS, ge=1)
|
|
51
|
+
max_env_actions: int = Field(default=DEFAULT_MAX_ENV_ACTIONS, ge=1)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class BudgetExceeded(RuntimeError):
|
|
55
|
+
"""Raised by the kit when harness code exhausts an episode budget."""
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class RuntimeKit:
|
|
59
|
+
"""Everything harness code may touch, and the recorder of what it actually did."""
|
|
60
|
+
|
|
61
|
+
def __init__(
|
|
62
|
+
self,
|
|
63
|
+
*,
|
|
64
|
+
task_id: str,
|
|
65
|
+
instruction: str,
|
|
66
|
+
environment: AgentEnvironment,
|
|
67
|
+
provider: Provider,
|
|
68
|
+
tools: list[ToolSpec],
|
|
69
|
+
skills: SkillLibrary,
|
|
70
|
+
temperature: float,
|
|
71
|
+
budget: RunBudget,
|
|
72
|
+
system_prompt: str = "",
|
|
73
|
+
) -> None:
|
|
74
|
+
self.task_id = task_id
|
|
75
|
+
self.instruction = instruction
|
|
76
|
+
self.temperature = temperature
|
|
77
|
+
self.tools = tools
|
|
78
|
+
# The doc's assembled prompt (prompt surfaces + tools + skills index): prompt surfaces
|
|
79
|
+
# stay meaningful alongside a code surface, and code may use or ignore this.
|
|
80
|
+
self.system_prompt = system_prompt
|
|
81
|
+
self._environment = environment
|
|
82
|
+
self._provider = provider
|
|
83
|
+
self._skills = skills
|
|
84
|
+
self._budget = budget
|
|
85
|
+
self._llm_calls = 0
|
|
86
|
+
self._env_actions = 0
|
|
87
|
+
self.steps: list[Step] = []
|
|
88
|
+
|
|
89
|
+
# -- the two budgeted primitives -----------------------------------------------------------
|
|
90
|
+
|
|
91
|
+
def complete(
|
|
92
|
+
self,
|
|
93
|
+
system: str,
|
|
94
|
+
messages: list[Message] | list[tuple[str, str]],
|
|
95
|
+
*,
|
|
96
|
+
temperature: float | None = None,
|
|
97
|
+
max_tokens: int = 2048,
|
|
98
|
+
) -> str:
|
|
99
|
+
"""One LLM call. Accepts `Message`s or plain `(role, content)` tuples."""
|
|
100
|
+
if self._llm_calls >= self._budget.max_llm_calls:
|
|
101
|
+
raise BudgetExceeded(f"llm call budget exhausted ({self._budget.max_llm_calls})")
|
|
102
|
+
self._llm_calls += 1
|
|
103
|
+
normalized = [_to_message(m) for m in messages]
|
|
104
|
+
completion = self._provider.complete(
|
|
105
|
+
system,
|
|
106
|
+
normalized,
|
|
107
|
+
temperature=self.temperature if temperature is None else temperature,
|
|
108
|
+
max_tokens=max_tokens,
|
|
109
|
+
)
|
|
110
|
+
return completion.text
|
|
111
|
+
|
|
112
|
+
def execute(self, tool: str, arguments: JsonObject) -> Observation:
|
|
113
|
+
"""One environment action, validated against the tool policy and recorded verbatim."""
|
|
114
|
+
if self._env_actions >= self._budget.max_env_actions:
|
|
115
|
+
raise BudgetExceeded(f"env action budget exhausted ({self._budget.max_env_actions})")
|
|
116
|
+
action = Action(kind=ActionKind.TOOL_CALL, name=tool, arguments=arguments)
|
|
117
|
+
if tool not in {t.name for t in self.tools} or not is_env_action(action):
|
|
118
|
+
observation = Observation(content=f"tool {tool!r} not available", is_error=True)
|
|
119
|
+
else:
|
|
120
|
+
self._env_actions += 1
|
|
121
|
+
observation = self._environment.execute(action)
|
|
122
|
+
self.steps.append(
|
|
123
|
+
Step(
|
|
124
|
+
action=action,
|
|
125
|
+
observation=observation,
|
|
126
|
+
state_before=EnvState(),
|
|
127
|
+
task=self.instruction,
|
|
128
|
+
)
|
|
129
|
+
)
|
|
130
|
+
return observation
|
|
131
|
+
|
|
132
|
+
# -- conveniences (unbudgeted, side-effect free) --------------------------------------------
|
|
133
|
+
|
|
134
|
+
def parse_tool_call(self, text: str) -> ToolCall | None:
|
|
135
|
+
return parse_tool_call(text)
|
|
136
|
+
|
|
137
|
+
def tools_text(self) -> str:
|
|
138
|
+
return render_tools(self.tools)
|
|
139
|
+
|
|
140
|
+
def skills_index(self) -> str:
|
|
141
|
+
return self._skills.render_index()
|
|
142
|
+
|
|
143
|
+
def read_skill(self, name: str) -> str | None:
|
|
144
|
+
skill = self._skills.get(name)
|
|
145
|
+
return skill.body if skill is not None else None
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _to_message(m: Message | tuple[str, str]) -> Message:
|
|
149
|
+
if isinstance(m, Message):
|
|
150
|
+
return m
|
|
151
|
+
role, content = m
|
|
152
|
+
if role == "user":
|
|
153
|
+
return Message(role="user", content=content)
|
|
154
|
+
if role == "assistant":
|
|
155
|
+
return Message(role="assistant", content=content)
|
|
156
|
+
raise ValueError(f"message role must be 'user' or 'assistant', got {role!r}")
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def compile_harness_code(code: str) -> None:
|
|
160
|
+
"""Front-loaded validation: the code must compile and define `run` at module scope.
|
|
161
|
+
|
|
162
|
+
Raises `ValueError` so `HarnessDoc` construction rejects an unrunnable harness before any
|
|
163
|
+
eval budget could be spent on it. Behavioral quality is the gate's job, not this one's.
|
|
164
|
+
"""
|
|
165
|
+
try:
|
|
166
|
+
compiled = compile(code, "<code:runtime>", "exec")
|
|
167
|
+
except SyntaxError as exc:
|
|
168
|
+
raise ValueError(f"code:runtime does not compile: {exc}") from exc
|
|
169
|
+
names = set(compiled.co_names)
|
|
170
|
+
# A module that never binds `run` can't be dispatched. co_names covers references, so also
|
|
171
|
+
# accept a top-level def by scanning consts for the code object.
|
|
172
|
+
defines_run = (
|
|
173
|
+
any(getattr(const, "co_name", None) == CODE_ENTRYPOINT for const in compiled.co_consts)
|
|
174
|
+
or CODE_ENTRYPOINT in names
|
|
175
|
+
)
|
|
176
|
+
if not defines_run:
|
|
177
|
+
raise ValueError(f"code:runtime must define `{CODE_ENTRYPOINT}(kit)` at module scope")
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
class CodeRuntime:
|
|
181
|
+
"""Drives episodes through a harness-defined `run(kit)` instead of the fixed loop."""
|
|
182
|
+
|
|
183
|
+
def __init__(
|
|
184
|
+
self,
|
|
185
|
+
provider: Provider,
|
|
186
|
+
*,
|
|
187
|
+
code: str,
|
|
188
|
+
tools: list[ToolSpec],
|
|
189
|
+
temperature: float = 0.7,
|
|
190
|
+
skills: SkillLibrary | None = None,
|
|
191
|
+
budget: RunBudget | None = None,
|
|
192
|
+
system_prompt: str = "",
|
|
193
|
+
) -> None:
|
|
194
|
+
compile_harness_code(code)
|
|
195
|
+
namespace: dict[str, object] = {"__name__": "wmo_harness_code"}
|
|
196
|
+
exec(code, namespace) # noqa: S102 - the code surface IS the artifact under search
|
|
197
|
+
entry = namespace.get(CODE_ENTRYPOINT)
|
|
198
|
+
if not callable(entry):
|
|
199
|
+
raise ValueError(f"code:runtime `{CODE_ENTRYPOINT}` is not callable")
|
|
200
|
+
self._run = cast("Callable[[RuntimeKit], object]", entry)
|
|
201
|
+
self._provider = provider
|
|
202
|
+
self._tools = tools
|
|
203
|
+
self._temperature = temperature
|
|
204
|
+
self._skills = skills if skills is not None else SkillLibrary()
|
|
205
|
+
self._budget = budget if budget is not None else RunBudget()
|
|
206
|
+
self._system_prompt = system_prompt
|
|
207
|
+
|
|
208
|
+
def run(self, task_id: str, instruction: str, environment: AgentEnvironment) -> RunResult:
|
|
209
|
+
kit = RuntimeKit(
|
|
210
|
+
task_id=task_id,
|
|
211
|
+
instruction=instruction,
|
|
212
|
+
environment=environment,
|
|
213
|
+
provider=self._provider,
|
|
214
|
+
tools=self._tools,
|
|
215
|
+
skills=self._skills,
|
|
216
|
+
temperature=self._temperature,
|
|
217
|
+
budget=self._budget,
|
|
218
|
+
system_prompt=self._system_prompt,
|
|
219
|
+
)
|
|
220
|
+
try:
|
|
221
|
+
answer = self._run(kit)
|
|
222
|
+
except BudgetExceeded as exc:
|
|
223
|
+
return self._result(kit, StopReason.BUDGET, note=str(exc))
|
|
224
|
+
except Exception as exc: # noqa: BLE001 - crash-isolation is the contract
|
|
225
|
+
return self._result(kit, StopReason.ERROR, note=f"{type(exc).__name__}: {exc}")
|
|
226
|
+
return RunResult(
|
|
227
|
+
task_id=kit.task_id,
|
|
228
|
+
steps=kit.steps,
|
|
229
|
+
stop_reason=StopReason.SUBMITTED,
|
|
230
|
+
answer=answer if isinstance(answer, str) else "",
|
|
231
|
+
turns=len(kit.steps),
|
|
232
|
+
)
|
|
233
|
+
|
|
234
|
+
def _result(self, kit: RuntimeKit, stop_reason: StopReason, *, note: str) -> RunResult:
|
|
235
|
+
# The kit-recorded partial transcript survives, plus one error step so the judge (and the
|
|
236
|
+
# failure clustering) can see WHY the episode ended.
|
|
237
|
+
kit.steps.append(
|
|
238
|
+
Step(
|
|
239
|
+
action=Action(kind=ActionKind.MESSAGE, content="(harness runtime)"),
|
|
240
|
+
observation=Observation(content=note, is_error=True),
|
|
241
|
+
state_before=EnvState(),
|
|
242
|
+
task=kit.instruction,
|
|
243
|
+
)
|
|
244
|
+
)
|
|
245
|
+
return RunResult(
|
|
246
|
+
task_id=kit.task_id,
|
|
247
|
+
steps=kit.steps,
|
|
248
|
+
stop_reason=stop_reason,
|
|
249
|
+
answer="",
|
|
250
|
+
turns=len(kit.steps),
|
|
251
|
+
)
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
# The reference loop, as the seed content of a `code:runtime` surface: functionally the fixed
|
|
255
|
+
# `AgentRuntime` loop, expressed through the kit so the search can restructure it.
|
|
256
|
+
DEFAULT_RUNTIME_CODE = '''"""Baseline agent loop: one JSON tool call per turn.
|
|
257
|
+
|
|
258
|
+
Calling `submit` ends the episode."""
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def run(kit):
|
|
262
|
+
messages = [("user", "TASK: " + kit.instruction)]
|
|
263
|
+
nudged = False
|
|
264
|
+
for _ in range(20):
|
|
265
|
+
reply = kit.complete(kit.system_prompt, messages).strip()
|
|
266
|
+
call = kit.parse_tool_call(reply)
|
|
267
|
+
if call is None:
|
|
268
|
+
if nudged:
|
|
269
|
+
return ""
|
|
270
|
+
nudged = True
|
|
271
|
+
messages.append(("assistant", reply))
|
|
272
|
+
messages.append(("user", "[ERROR] reply with EXACTLY one JSON object: "
|
|
273
|
+
"{\\"tool\\": \\"<name>\\", \\"arguments\\": {...}}"))
|
|
274
|
+
continue
|
|
275
|
+
if call.tool == "submit":
|
|
276
|
+
answer = call.arguments.get("answer")
|
|
277
|
+
return answer if isinstance(answer, str) else ""
|
|
278
|
+
if call.tool == "read_skill":
|
|
279
|
+
name = call.arguments.get("name")
|
|
280
|
+
body = kit.read_skill(name if isinstance(name, str) else "")
|
|
281
|
+
text, is_error = (body, False) if body is not None else ("no such skill", True)
|
|
282
|
+
else:
|
|
283
|
+
observation = kit.execute(call.tool, call.arguments)
|
|
284
|
+
text, is_error = observation.content, observation.is_error
|
|
285
|
+
messages.append(("assistant", reply))
|
|
286
|
+
messages.append(("user", ("[ERROR] " if is_error else "[OK] ") + text))
|
|
287
|
+
return ""
|
|
288
|
+
'''
|