world-model-optimizer 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_waterfall/LICENSE +21 -0
- llm_waterfall/__init__.py +53 -0
- llm_waterfall/adapters/__init__.py +36 -0
- llm_waterfall/adapters/anthropic.py +105 -0
- llm_waterfall/adapters/aws_mantle.py +47 -0
- llm_waterfall/adapters/azure_openai.py +71 -0
- llm_waterfall/adapters/base.py +51 -0
- llm_waterfall/adapters/bedrock.py +309 -0
- llm_waterfall/adapters/openai.py +130 -0
- llm_waterfall/classify.py +184 -0
- llm_waterfall/pricing.py +110 -0
- llm_waterfall/py.typed +0 -0
- llm_waterfall/types.py +295 -0
- llm_waterfall/waterfall.py +255 -0
- wmo/__init__.py +38 -0
- wmo/agents/__init__.py +7 -0
- wmo/agents/default.py +29 -0
- wmo/agents/meta.py +55 -0
- wmo/agents/optimizer.py +55 -0
- wmo/agents/project.py +928 -0
- wmo/cli/__init__.py +5 -0
- wmo/cli/agent_session.py +1123 -0
- wmo/cli/app.py +2489 -0
- wmo/cli/e2b_cmds.py +212 -0
- wmo/cli/eval_closed_loop.py +207 -0
- wmo/cli/harness_app.py +1147 -0
- wmo/cli/harness_distill.py +659 -0
- wmo/cli/hosted_session.py +880 -0
- wmo/cli/ingest_cmd.py +165 -0
- wmo/cli/model_roles.py +82 -0
- wmo/cli/platform_cmds.py +372 -0
- wmo/cli/route_app.py +274 -0
- wmo/cli/session_state.py +243 -0
- wmo/cli/ui.py +1107 -0
- wmo/cli/workspace_sync.py +504 -0
- wmo/config/__init__.py +60 -0
- wmo/config/card.py +129 -0
- wmo/config/config.py +367 -0
- wmo/config/dotenv.py +67 -0
- wmo/config/settings.py +128 -0
- wmo/config/store.py +177 -0
- wmo/conftest.py +19 -0
- wmo/connect/__init__.py +88 -0
- wmo/connect/apps.py +78 -0
- wmo/connect/brave.py +284 -0
- wmo/connect/connector.py +79 -0
- wmo/connect/credentials.py +164 -0
- wmo/connect/github.py +321 -0
- wmo/connect/google.py +627 -0
- wmo/connect/notion.py +790 -0
- wmo/connect/oauth.py +461 -0
- wmo/connect/slack.py +555 -0
- wmo/connect/store.py +199 -0
- wmo/connect/types.py +156 -0
- wmo/core/__init__.py +21 -0
- wmo/core/parsing.py +281 -0
- wmo/core/render.py +271 -0
- wmo/core/text.py +40 -0
- wmo/core/types.py +116 -0
- wmo/distill/__init__.py +14 -0
- wmo/distill/agents.py +140 -0
- wmo/distill/config.py +1006 -0
- wmo/distill/cost.py +437 -0
- wmo/distill/data.py +921 -0
- wmo/distill/deadlines.py +254 -0
- wmo/distill/fake_tinker.py +734 -0
- wmo/distill/gate.py +122 -0
- wmo/distill/loop.py +3499 -0
- wmo/distill/renderers.py +399 -0
- wmo/distill/rendering.py +620 -0
- wmo/distill/rollouts.py +726 -0
- wmo/distill/samples.py +195 -0
- wmo/distill/store.py +829 -0
- wmo/distill/teacher.py +714 -0
- wmo/distill/tokens.py +535 -0
- wmo/distill/tracking.py +552 -0
- wmo/distill/tripwire.py +411 -0
- wmo/distill/xtoken/byte_offsets.py +152 -0
- wmo/distill/xtoken/chunks.py +457 -0
- wmo/distill/xtoken/prompt_logprobs.py +475 -0
- wmo/distill/xtoken/teacher_render.py +346 -0
- wmo/engine/__init__.py +28 -0
- wmo/engine/autoconfig.py +367 -0
- wmo/engine/build.py +346 -0
- wmo/engine/demo.py +77 -0
- wmo/engine/eval_suites.py +245 -0
- wmo/engine/grounding.py +491 -0
- wmo/engine/knowledge.py +291 -0
- wmo/engine/loader.py +36 -0
- wmo/engine/play.py +92 -0
- wmo/engine/prompts.py +99 -0
- wmo/engine/replay.py +443 -0
- wmo/engine/reporting.py +58 -0
- wmo/engine/workspace.py +468 -0
- wmo/engine/world_model.py +568 -0
- wmo/env/__init__.py +22 -0
- wmo/env/base.py +121 -0
- wmo/env/closed_loop.py +229 -0
- wmo/env/episode.py +107 -0
- wmo/env/llm_agent.py +93 -0
- wmo/env/scenarios.py +73 -0
- wmo/evals/__init__.py +52 -0
- wmo/evals/agreement.py +110 -0
- wmo/evals/base.py +45 -0
- wmo/evals/closed_loop.py +480 -0
- wmo/evals/failover.py +96 -0
- wmo/evals/gold.py +127 -0
- wmo/evals/grid.py +394 -0
- wmo/evals/grid_plot.py +205 -0
- wmo/evals/harbor/__init__.py +27 -0
- wmo/evals/harbor/agent.py +573 -0
- wmo/evals/harbor/ctrf.py +171 -0
- wmo/evals/harbor/e2b_environment.py +587 -0
- wmo/evals/harbor/e2b_template_policy.py +144 -0
- wmo/evals/harbor/scorer.py +875 -0
- wmo/evals/harbor/tasks.py +140 -0
- wmo/evals/open_loop.py +194 -0
- wmo/evals/tasks.py +53 -0
- wmo/harness/__init__.py +51 -0
- wmo/harness/code_runtime.py +288 -0
- wmo/harness/create.py +1191 -0
- wmo/harness/delta.py +220 -0
- wmo/harness/doc.py +556 -0
- wmo/harness/e2b_ledger.py +342 -0
- wmo/harness/e2b_reap.py +476 -0
- wmo/harness/e2b_sandbox.py +350 -0
- wmo/harness/environment.py +35 -0
- wmo/harness/live_session.py +543 -0
- wmo/harness/mutate.py +343 -0
- wmo/harness/pi_e2b.py +1710 -0
- wmo/harness/pi_entry/entry.ts +268 -0
- wmo/harness/pi_entry/runner_frames.ts +92 -0
- wmo/harness/pi_entry/runner_live.ts +587 -0
- wmo/harness/pi_entry/runner_service.ts +270 -0
- wmo/harness/pi_entry/runner_stdio.ts +374 -0
- wmo/harness/pi_entry/runner_termination.ts +142 -0
- wmo/harness/pi_local.py +262 -0
- wmo/harness/pi_runtime.py +495 -0
- wmo/harness/pi_vendor.py +65 -0
- wmo/harness/population.py +509 -0
- wmo/harness/project_proposer.py +569 -0
- wmo/harness/proposer.py +977 -0
- wmo/harness/runner_link.py +619 -0
- wmo/harness/runtime.py +389 -0
- wmo/harness/scoring.py +247 -0
- wmo/harness/skills.py +116 -0
- wmo/harness/source_tree.py +319 -0
- wmo/harness/store.py +176 -0
- wmo/harness/tools.py +105 -0
- wmo/harness/vendor/manifest.sha256 +58 -0
- wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
- wmo/harness/vendor/pi-agent/LICENSE +21 -0
- wmo/harness/vendor/pi-agent/README.md +488 -0
- wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
- wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
- wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
- wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
- wmo/harness/vendor/pi-agent/docs/models.md +966 -0
- wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
- wmo/harness/vendor/pi-agent/package.json +60 -0
- wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
- wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
- wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
- wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
- wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
- wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
- wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
- wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
- wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
- wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
- wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
- wmo/harness/vendor/pi-agent/src/index.ts +44 -0
- wmo/harness/vendor/pi-agent/src/node.ts +2 -0
- wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
- wmo/harness/vendor/pi-agent/src/types.ts +428 -0
- wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
- wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
- wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
- wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
- wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
- wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
- wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
- wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
- wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
- wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
- wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
- wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
- wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
- wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
- wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
- wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
- wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
- wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
- wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
- wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
- wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
- wmo/harness/vendor/vendor_pi.sh +59 -0
- wmo/harness/workspace_patch.py +270 -0
- wmo/ingest/__init__.py +47 -0
- wmo/ingest/adapter.py +72 -0
- wmo/ingest/base.py +114 -0
- wmo/ingest/braintrust.py +339 -0
- wmo/ingest/detect.py +126 -0
- wmo/ingest/langfuse.py +291 -0
- wmo/ingest/langsmith.py +444 -0
- wmo/ingest/mastra.py +330 -0
- wmo/ingest/messages.py +170 -0
- wmo/ingest/normalize.py +679 -0
- wmo/ingest/otel_genai.py +69 -0
- wmo/ingest/otel_writer.py +100 -0
- wmo/ingest/phoenix.py +150 -0
- wmo/ingest/postgres.py +246 -0
- wmo/ingest/posthog.py +320 -0
- wmo/ingest/quality.py +28 -0
- wmo/ingest/stream.py +209 -0
- wmo/ingest/testdata/sample_otlp.json +60 -0
- wmo/ingest/testdata/sample_spans.jsonl +3 -0
- wmo/optimize/__init__.py +25 -0
- wmo/optimize/base.py +143 -0
- wmo/optimize/gepa.py +806 -0
- wmo/optimize/judge.py +262 -0
- wmo/optimize/judge_quality.py +359 -0
- wmo/optimize/knn.py +468 -0
- wmo/optimize/numeric.py +152 -0
- wmo/optimize/outcomes.py +103 -0
- wmo/optimize/policy.py +669 -0
- wmo/optimize/report.py +231 -0
- wmo/optimize/reward.py +129 -0
- wmo/optimize/routing.py +373 -0
- wmo/platform/__init__.py +6 -0
- wmo/platform/auth.py +115 -0
- wmo/platform/client.py +551 -0
- wmo/platform/credentials.py +126 -0
- wmo/platform/transfer.py +158 -0
- wmo/providers/__init__.py +40 -0
- wmo/providers/_bedrock_chat.py +155 -0
- wmo/providers/_openai_common.py +182 -0
- wmo/providers/_responses_common.py +472 -0
- wmo/providers/anthropic.py +134 -0
- wmo/providers/azure_openai.py +296 -0
- wmo/providers/base.py +300 -0
- wmo/providers/bedrock.py +312 -0
- wmo/providers/models.py +205 -0
- wmo/providers/openai.py +143 -0
- wmo/providers/openai_responses.py +240 -0
- wmo/providers/pool.py +170 -0
- wmo/providers/registry.py +73 -0
- wmo/providers/retry.py +151 -0
- wmo/providers/tinker.py +936 -0
- wmo/providers/waterfall.py +336 -0
- wmo/research/__init__.py +81 -0
- wmo/research/ablation.py +133 -0
- wmo/research/concurrency_plot.py +523 -0
- wmo/research/concurrency_run.py +240 -0
- wmo/research/concurrency_scaling.py +270 -0
- wmo/research/gepa_scaling.py +274 -0
- wmo/research/pipeline.py +198 -0
- wmo/research/scaling_split.py +82 -0
- wmo/research/scenario_fidelity.py +198 -0
- wmo/research/scenario_recovery.py +92 -0
- wmo/research/seed_stability.py +90 -0
- wmo/research/trace_scaling.py +348 -0
- wmo/retrieval/__init__.py +6 -0
- wmo/retrieval/embedders.py +105 -0
- wmo/retrieval/leakfree.py +52 -0
- wmo/retrieval/retriever.py +173 -0
- wmo/scenarios/__init__.py +58 -0
- wmo/scenarios/builder.py +152 -0
- wmo/scenarios/mining/__init__.py +27 -0
- wmo/scenarios/mining/clustering.py +171 -0
- wmo/scenarios/mining/facets.py +226 -0
- wmo/scenarios/mining/selection.py +220 -0
- wmo/scenarios/synthesis/__init__.py +6 -0
- wmo/scenarios/synthesis/scenario_set.py +63 -0
- wmo/scenarios/synthesis/synthesizer.py +85 -0
- wmo/scenarios/verification/__init__.py +17 -0
- wmo/scenarios/verification/judge.py +97 -0
- wmo/scenarios/verification/verify.py +135 -0
- wmo/serving/__init__.py +5 -0
- wmo/serving/builds.py +451 -0
- wmo/serving/chat.py +878 -0
- wmo/serving/endpoint_config.py +64 -0
- wmo/serving/savings.py +250 -0
- wmo/serving/server.py +553 -0
- wmo/serving/traces_source.py +206 -0
- wmo/telemetry.py +213 -0
- wmo/tracking/__init__.py +36 -0
- wmo/tracking/clock.py +24 -0
- wmo/tracking/metered.py +125 -0
- wmo/tracking/pricing.py +99 -0
- wmo/tracking/store.py +31 -0
- wmo/tracking/tracker.py +149 -0
- world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
- world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
- world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
- world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
wmo/__init__.py
ADDED
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""World Model Optimizer — a frontier LLM acts as your agent's environment.
|
|
2
|
+
|
|
3
|
+
Public API:
|
|
4
|
+
from wmo import WorldModel
|
|
5
|
+
wm = WorldModel.load(".wmo", provider=...)
|
|
6
|
+
session = wm.new_session(task="browse the shop")
|
|
7
|
+
obs = wm.step(session.id, action)
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from wmo.core.types import (
|
|
11
|
+
Action,
|
|
12
|
+
ActionKind,
|
|
13
|
+
EnvState,
|
|
14
|
+
Observation,
|
|
15
|
+
Session,
|
|
16
|
+
Step,
|
|
17
|
+
Trace,
|
|
18
|
+
)
|
|
19
|
+
from wmo.engine.world_model import WorldModel
|
|
20
|
+
from wmo.env import DONE_SIGNAL, Agent, Env, EpisodeResult, StopReason, WorldModelEnv, run_episode
|
|
21
|
+
|
|
22
|
+
__all__ = [
|
|
23
|
+
"WorldModel",
|
|
24
|
+
"Action",
|
|
25
|
+
"ActionKind",
|
|
26
|
+
"Observation",
|
|
27
|
+
"EnvState",
|
|
28
|
+
"Session",
|
|
29
|
+
"Step",
|
|
30
|
+
"Trace",
|
|
31
|
+
"Agent",
|
|
32
|
+
"DONE_SIGNAL",
|
|
33
|
+
"Env",
|
|
34
|
+
"EpisodeResult",
|
|
35
|
+
"StopReason",
|
|
36
|
+
"WorldModelEnv",
|
|
37
|
+
"run_episode",
|
|
38
|
+
]
|
wmo/agents/__init__.py
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
"""Agent definitions and project-backed session execution."""
|
|
2
|
+
|
|
3
|
+
from wmo.agents.default import default_agent
|
|
4
|
+
from wmo.agents.meta import meta_agent
|
|
5
|
+
from wmo.agents.project import AgentProject, AgentProjectRun
|
|
6
|
+
|
|
7
|
+
__all__ = ["AgentProject", "AgentProjectRun", "default_agent", "meta_agent"]
|
wmo/agents/default.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""The default pi agent definition shipped by wmo."""
|
|
2
|
+
|
|
3
|
+
from wmo.harness.doc import (
|
|
4
|
+
MAX_OUTPUT_TOKENS_ID,
|
|
5
|
+
RUNTIME_KIND_ID,
|
|
6
|
+
HarnessDoc,
|
|
7
|
+
Surface,
|
|
8
|
+
SurfaceKind,
|
|
9
|
+
)
|
|
10
|
+
from wmo.harness.pi_vendor import pi_agent_code_surfaces
|
|
11
|
+
from wmo.harness.runtime import DEFAULT_MAX_OUTPUT_TOKENS
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def default_agent(name: str = "default") -> HarnessDoc:
|
|
15
|
+
"""Return an independent default-agent document backed by vendored pi."""
|
|
16
|
+
base = HarnessDoc.baseline(name)
|
|
17
|
+
return HarnessDoc(
|
|
18
|
+
name=name,
|
|
19
|
+
surfaces=[
|
|
20
|
+
*base.surfaces,
|
|
21
|
+
Surface(
|
|
22
|
+
id=MAX_OUTPUT_TOKENS_ID,
|
|
23
|
+
kind=SurfaceKind.PARAM,
|
|
24
|
+
content=str(DEFAULT_MAX_OUTPUT_TOKENS),
|
|
25
|
+
),
|
|
26
|
+
Surface(id=RUNTIME_KIND_ID, kind=SurfaceKind.PARAM, content="pi-node"),
|
|
27
|
+
*pi_agent_code_surfaces(),
|
|
28
|
+
],
|
|
29
|
+
)
|
wmo/agents/meta.py
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""The project agent that proposes harness improvements."""
|
|
2
|
+
|
|
3
|
+
from wmo.agents.default import default_agent
|
|
4
|
+
from wmo.harness.doc import MAX_OUTPUT_TOKENS_ID, MAX_TURNS_ID, TOOL_POLICY_ID, HarnessDoc
|
|
5
|
+
|
|
6
|
+
META_AGENT_PROMPT = """You are the meta agent inside an optimizer project. Improve agent harnesses;
|
|
7
|
+
do not solve their benchmark tasks yourself.
|
|
8
|
+
|
|
9
|
+
The project filesystem is your durable memory. Each iteration provides a current parent document,
|
|
10
|
+
failure evidence, and the complete judged history. Earlier proposal files remain under proposals/.
|
|
11
|
+
Parent/evidence/history manifests point to bounded content files; read those files selectively and
|
|
12
|
+
follow their exact paths with read_file. Treat context/, evaluations/, and earlier proposals as
|
|
13
|
+
immutable evidence; use write_file only for every required proposal output. Read the selected
|
|
14
|
+
failure's execution traces and judge reasons, not every available file.
|
|
15
|
+
Every project turn has a bounded tool/turn budget. Within the first 12 read_file calls, write a
|
|
16
|
+
complete, parseable draft to every required proposal output. Those files are durable checkpoints:
|
|
17
|
+
keep them valid while using remaining actions to inspect targeted source/evidence and refine them.
|
|
18
|
+
Never spend the whole turn exploring before writing. On a repair turn, read the validation report
|
|
19
|
+
and rewrite every invalid slot before any optional exploration.
|
|
20
|
+
Distinguish a harness failure from an unavailable or mis-simulated environment: do not spend
|
|
21
|
+
another proposal merely retrying an unreachable endpoint.
|
|
22
|
+
Inspect earlier proposals and evaluations, learn from accepted and rejected attempts, and produce
|
|
23
|
+
the exact number of independent proposals requested for the iteration.
|
|
24
|
+
|
|
25
|
+
The harness's real source-code surfaces are the primary search space. Prefer a focused structural
|
|
26
|
+
code change when the failure is in control flow, context handling, tool dispatch, verification,
|
|
27
|
+
recovery, or output parsing. Use a skill for a reusable technique, tool policy for capability,
|
|
28
|
+
and params for genuine sampling/budget issues. Prompt wording is the weakest lever. Every proposal
|
|
29
|
+
must target the supplied parent, change one mechanism, preserve unrelated behavior, and state a
|
|
30
|
+
falsifiable expected effect. Compact exact edits are preferred for large source files. Never
|
|
31
|
+
overwrite an earlier iteration.
|
|
32
|
+
|
|
33
|
+
Use read_file and write_file to work in the project. The user message for each iteration gives the
|
|
34
|
+
required input and output paths and the proposal schema. Write every requested proposal before
|
|
35
|
+
calling submit. Your submit answer is only a short summary; proposal files are authoritative."""
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def meta_agent(name: str = "meta") -> HarnessDoc:
|
|
39
|
+
"""Return the meta-agent document as a separate pi-derived agent."""
|
|
40
|
+
base = default_agent(name)
|
|
41
|
+
surfaces = []
|
|
42
|
+
for surface in base.surfaces:
|
|
43
|
+
if surface.id == "prompt:core":
|
|
44
|
+
surfaces.append(surface.model_copy(update={"content": META_AGENT_PROMPT}))
|
|
45
|
+
elif surface.id == TOOL_POLICY_ID:
|
|
46
|
+
surfaces.append(surface.model_copy(update={"content": "read_file\nwrite_file\nsubmit"}))
|
|
47
|
+
elif surface.id == MAX_TURNS_ID:
|
|
48
|
+
surfaces.append(surface.model_copy(update={"content": "60"}))
|
|
49
|
+
elif surface.id == MAX_OUTPUT_TOKENS_ID:
|
|
50
|
+
# GPT-5.5 high reasoning spends output tokens before its visible filesystem calls. A
|
|
51
|
+
# batch of three compact proposals needs the same 16k headroom as the direct proposer.
|
|
52
|
+
surfaces.append(surface.model_copy(update={"content": "16384"}))
|
|
53
|
+
else:
|
|
54
|
+
surfaces.append(surface)
|
|
55
|
+
return HarnessDoc(name=name, surfaces=surfaces)
|
wmo/agents/optimizer.py
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""The built-in project agent that proposes complete harness source trees."""
|
|
2
|
+
|
|
3
|
+
from wmo.agents.default import default_agent
|
|
4
|
+
from wmo.harness.doc import MAX_OUTPUT_TOKENS_ID, MAX_TURNS_ID, TOOL_POLICY_ID, HarnessDoc
|
|
5
|
+
|
|
6
|
+
OPTIMIZER_AGENT_PROMPT = """You are an optimization agent inside a harness project.
|
|
7
|
+
Improve complete harness source trees; do not solve their evaluation tasks yourself.
|
|
8
|
+
|
|
9
|
+
The project filesystem is your evidence. It contains every earlier complete source tree, its full
|
|
10
|
+
score report, raw per-trial evaluator artifacts, and previous proposal traces. Read the history
|
|
11
|
+
manifest and inspect the most relevant raw files before deciding what to change. Treat all history
|
|
12
|
+
and proposal records as immutable evidence.
|
|
13
|
+
|
|
14
|
+
Each project request names one preinitialized output directory containing a complete starting
|
|
15
|
+
source tree. Work only there. You may edit, delete, or replace any files, copy mechanisms from
|
|
16
|
+
earlier source trees, combine several mechanisms, or build a new tree, but the final directory must
|
|
17
|
+
stand alone.
|
|
18
|
+
|
|
19
|
+
Propose general-purpose harness mechanisms for the task distribution, not solutions or hints for
|
|
20
|
+
particular evaluation instances. Never hard-code or copy literal instance names or identifiers,
|
|
21
|
+
instance-specific strings, expected answers, fixture details, or special-case branches recognizable
|
|
22
|
+
as targeting one instance into any candidate path, filename, source file, prompt, comment, skill,
|
|
23
|
+
configuration, or test. General mechanisms inferred from prior evidence are allowed only when they
|
|
24
|
+
would be useful across many unfamiliar tasks. Subject to that constraint, the complete portable
|
|
25
|
+
source remains freely rewritable, including its control flow, tools, prompts, model-call strategy,
|
|
26
|
+
runtime code, and configuration. This is not a patch-only search.
|
|
27
|
+
|
|
28
|
+
Candidate filenames follow a strict grammar. Outside the reserved names (`SYSTEM.md`,
|
|
29
|
+
`config.toml`, `runtime.py`, `skills/<skill-name>.md`), every path must be lowercase kebab-case:
|
|
30
|
+
runs of [a-z0-9] separated by single '/', '.', or '-' characters (for example
|
|
31
|
+
`src/agent-loop.ts`). Uppercase letters and underscores are rejected, paths that differ only in
|
|
32
|
+
letter case or only by '/' versus '.' collide, and no file path may also be a directory prefix of
|
|
33
|
+
another. One bad filename invalidates the whole candidate.
|
|
34
|
+
|
|
35
|
+
Inspect and test your work in the output directory. Do not write candidate files anywhere else.
|
|
36
|
+
Call submit only after the output directory contains the complete candidate requested by the host.
|
|
37
|
+
There is no repair turn, so leave a usable candidate on the first pass."""
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def optimizer_agent(name: str = "optimizer") -> HarnessDoc:
|
|
41
|
+
"""Return a pi-derived coding agent constrained to one project candidate stage."""
|
|
42
|
+
base = default_agent(name)
|
|
43
|
+
surfaces = []
|
|
44
|
+
for surface in base.surfaces:
|
|
45
|
+
if surface.id == "prompt:core":
|
|
46
|
+
surfaces.append(surface.model_copy(update={"content": OPTIMIZER_AGENT_PROMPT}))
|
|
47
|
+
elif surface.id == TOOL_POLICY_ID:
|
|
48
|
+
surfaces.append(surface.model_copy(update={"content": "bash\nread_file\nsubmit"}))
|
|
49
|
+
elif surface.id == MAX_TURNS_ID:
|
|
50
|
+
surfaces.append(surface.model_copy(update={"content": "60"}))
|
|
51
|
+
elif surface.id == MAX_OUTPUT_TOKENS_ID:
|
|
52
|
+
surfaces.append(surface.model_copy(update={"content": "16384"}))
|
|
53
|
+
else:
|
|
54
|
+
surfaces.append(surface)
|
|
55
|
+
return HarnessDoc(name=name, surfaces=surfaces)
|