world-model-optimizer 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_waterfall/LICENSE +21 -0
- llm_waterfall/__init__.py +53 -0
- llm_waterfall/adapters/__init__.py +36 -0
- llm_waterfall/adapters/anthropic.py +105 -0
- llm_waterfall/adapters/aws_mantle.py +47 -0
- llm_waterfall/adapters/azure_openai.py +71 -0
- llm_waterfall/adapters/base.py +51 -0
- llm_waterfall/adapters/bedrock.py +309 -0
- llm_waterfall/adapters/openai.py +130 -0
- llm_waterfall/classify.py +184 -0
- llm_waterfall/pricing.py +110 -0
- llm_waterfall/py.typed +0 -0
- llm_waterfall/types.py +295 -0
- llm_waterfall/waterfall.py +255 -0
- wmo/__init__.py +38 -0
- wmo/agents/__init__.py +7 -0
- wmo/agents/default.py +29 -0
- wmo/agents/meta.py +55 -0
- wmo/agents/optimizer.py +55 -0
- wmo/agents/project.py +928 -0
- wmo/cli/__init__.py +5 -0
- wmo/cli/agent_session.py +1123 -0
- wmo/cli/app.py +2489 -0
- wmo/cli/e2b_cmds.py +212 -0
- wmo/cli/eval_closed_loop.py +207 -0
- wmo/cli/harness_app.py +1147 -0
- wmo/cli/harness_distill.py +659 -0
- wmo/cli/hosted_session.py +880 -0
- wmo/cli/ingest_cmd.py +165 -0
- wmo/cli/model_roles.py +82 -0
- wmo/cli/platform_cmds.py +372 -0
- wmo/cli/route_app.py +274 -0
- wmo/cli/session_state.py +243 -0
- wmo/cli/ui.py +1107 -0
- wmo/cli/workspace_sync.py +504 -0
- wmo/config/__init__.py +60 -0
- wmo/config/card.py +129 -0
- wmo/config/config.py +367 -0
- wmo/config/dotenv.py +67 -0
- wmo/config/settings.py +128 -0
- wmo/config/store.py +177 -0
- wmo/conftest.py +19 -0
- wmo/connect/__init__.py +88 -0
- wmo/connect/apps.py +78 -0
- wmo/connect/brave.py +284 -0
- wmo/connect/connector.py +79 -0
- wmo/connect/credentials.py +164 -0
- wmo/connect/github.py +321 -0
- wmo/connect/google.py +627 -0
- wmo/connect/notion.py +790 -0
- wmo/connect/oauth.py +461 -0
- wmo/connect/slack.py +555 -0
- wmo/connect/store.py +199 -0
- wmo/connect/types.py +156 -0
- wmo/core/__init__.py +21 -0
- wmo/core/parsing.py +281 -0
- wmo/core/render.py +271 -0
- wmo/core/text.py +40 -0
- wmo/core/types.py +116 -0
- wmo/distill/__init__.py +14 -0
- wmo/distill/agents.py +140 -0
- wmo/distill/config.py +1006 -0
- wmo/distill/cost.py +437 -0
- wmo/distill/data.py +921 -0
- wmo/distill/deadlines.py +254 -0
- wmo/distill/fake_tinker.py +734 -0
- wmo/distill/gate.py +122 -0
- wmo/distill/loop.py +3499 -0
- wmo/distill/renderers.py +399 -0
- wmo/distill/rendering.py +620 -0
- wmo/distill/rollouts.py +726 -0
- wmo/distill/samples.py +195 -0
- wmo/distill/store.py +829 -0
- wmo/distill/teacher.py +714 -0
- wmo/distill/tokens.py +535 -0
- wmo/distill/tracking.py +552 -0
- wmo/distill/tripwire.py +411 -0
- wmo/distill/xtoken/byte_offsets.py +152 -0
- wmo/distill/xtoken/chunks.py +457 -0
- wmo/distill/xtoken/prompt_logprobs.py +475 -0
- wmo/distill/xtoken/teacher_render.py +346 -0
- wmo/engine/__init__.py +28 -0
- wmo/engine/autoconfig.py +367 -0
- wmo/engine/build.py +346 -0
- wmo/engine/demo.py +77 -0
- wmo/engine/eval_suites.py +245 -0
- wmo/engine/grounding.py +491 -0
- wmo/engine/knowledge.py +291 -0
- wmo/engine/loader.py +36 -0
- wmo/engine/play.py +92 -0
- wmo/engine/prompts.py +99 -0
- wmo/engine/replay.py +443 -0
- wmo/engine/reporting.py +58 -0
- wmo/engine/workspace.py +468 -0
- wmo/engine/world_model.py +568 -0
- wmo/env/__init__.py +22 -0
- wmo/env/base.py +121 -0
- wmo/env/closed_loop.py +229 -0
- wmo/env/episode.py +107 -0
- wmo/env/llm_agent.py +93 -0
- wmo/env/scenarios.py +73 -0
- wmo/evals/__init__.py +52 -0
- wmo/evals/agreement.py +110 -0
- wmo/evals/base.py +45 -0
- wmo/evals/closed_loop.py +480 -0
- wmo/evals/failover.py +96 -0
- wmo/evals/gold.py +127 -0
- wmo/evals/grid.py +394 -0
- wmo/evals/grid_plot.py +205 -0
- wmo/evals/harbor/__init__.py +27 -0
- wmo/evals/harbor/agent.py +573 -0
- wmo/evals/harbor/ctrf.py +171 -0
- wmo/evals/harbor/e2b_environment.py +587 -0
- wmo/evals/harbor/e2b_template_policy.py +144 -0
- wmo/evals/harbor/scorer.py +875 -0
- wmo/evals/harbor/tasks.py +140 -0
- wmo/evals/open_loop.py +194 -0
- wmo/evals/tasks.py +53 -0
- wmo/harness/__init__.py +51 -0
- wmo/harness/code_runtime.py +288 -0
- wmo/harness/create.py +1191 -0
- wmo/harness/delta.py +220 -0
- wmo/harness/doc.py +556 -0
- wmo/harness/e2b_ledger.py +342 -0
- wmo/harness/e2b_reap.py +476 -0
- wmo/harness/e2b_sandbox.py +350 -0
- wmo/harness/environment.py +35 -0
- wmo/harness/live_session.py +543 -0
- wmo/harness/mutate.py +343 -0
- wmo/harness/pi_e2b.py +1710 -0
- wmo/harness/pi_entry/entry.ts +268 -0
- wmo/harness/pi_entry/runner_frames.ts +92 -0
- wmo/harness/pi_entry/runner_live.ts +587 -0
- wmo/harness/pi_entry/runner_service.ts +270 -0
- wmo/harness/pi_entry/runner_stdio.ts +374 -0
- wmo/harness/pi_entry/runner_termination.ts +142 -0
- wmo/harness/pi_local.py +262 -0
- wmo/harness/pi_runtime.py +495 -0
- wmo/harness/pi_vendor.py +65 -0
- wmo/harness/population.py +509 -0
- wmo/harness/project_proposer.py +569 -0
- wmo/harness/proposer.py +977 -0
- wmo/harness/runner_link.py +619 -0
- wmo/harness/runtime.py +389 -0
- wmo/harness/scoring.py +247 -0
- wmo/harness/skills.py +116 -0
- wmo/harness/source_tree.py +319 -0
- wmo/harness/store.py +176 -0
- wmo/harness/tools.py +105 -0
- wmo/harness/vendor/manifest.sha256 +58 -0
- wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
- wmo/harness/vendor/pi-agent/LICENSE +21 -0
- wmo/harness/vendor/pi-agent/README.md +488 -0
- wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
- wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
- wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
- wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
- wmo/harness/vendor/pi-agent/docs/models.md +966 -0
- wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
- wmo/harness/vendor/pi-agent/package.json +60 -0
- wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
- wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
- wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
- wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
- wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
- wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
- wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
- wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
- wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
- wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
- wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
- wmo/harness/vendor/pi-agent/src/index.ts +44 -0
- wmo/harness/vendor/pi-agent/src/node.ts +2 -0
- wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
- wmo/harness/vendor/pi-agent/src/types.ts +428 -0
- wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
- wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
- wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
- wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
- wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
- wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
- wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
- wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
- wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
- wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
- wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
- wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
- wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
- wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
- wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
- wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
- wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
- wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
- wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
- wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
- wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
- wmo/harness/vendor/vendor_pi.sh +59 -0
- wmo/harness/workspace_patch.py +270 -0
- wmo/ingest/__init__.py +47 -0
- wmo/ingest/adapter.py +72 -0
- wmo/ingest/base.py +114 -0
- wmo/ingest/braintrust.py +339 -0
- wmo/ingest/detect.py +126 -0
- wmo/ingest/langfuse.py +291 -0
- wmo/ingest/langsmith.py +444 -0
- wmo/ingest/mastra.py +330 -0
- wmo/ingest/messages.py +170 -0
- wmo/ingest/normalize.py +679 -0
- wmo/ingest/otel_genai.py +69 -0
- wmo/ingest/otel_writer.py +100 -0
- wmo/ingest/phoenix.py +150 -0
- wmo/ingest/postgres.py +246 -0
- wmo/ingest/posthog.py +320 -0
- wmo/ingest/quality.py +28 -0
- wmo/ingest/stream.py +209 -0
- wmo/ingest/testdata/sample_otlp.json +60 -0
- wmo/ingest/testdata/sample_spans.jsonl +3 -0
- wmo/optimize/__init__.py +25 -0
- wmo/optimize/base.py +143 -0
- wmo/optimize/gepa.py +806 -0
- wmo/optimize/judge.py +262 -0
- wmo/optimize/judge_quality.py +359 -0
- wmo/optimize/knn.py +468 -0
- wmo/optimize/numeric.py +152 -0
- wmo/optimize/outcomes.py +103 -0
- wmo/optimize/policy.py +669 -0
- wmo/optimize/report.py +231 -0
- wmo/optimize/reward.py +129 -0
- wmo/optimize/routing.py +373 -0
- wmo/platform/__init__.py +6 -0
- wmo/platform/auth.py +115 -0
- wmo/platform/client.py +551 -0
- wmo/platform/credentials.py +126 -0
- wmo/platform/transfer.py +158 -0
- wmo/providers/__init__.py +40 -0
- wmo/providers/_bedrock_chat.py +155 -0
- wmo/providers/_openai_common.py +182 -0
- wmo/providers/_responses_common.py +472 -0
- wmo/providers/anthropic.py +134 -0
- wmo/providers/azure_openai.py +296 -0
- wmo/providers/base.py +300 -0
- wmo/providers/bedrock.py +312 -0
- wmo/providers/models.py +205 -0
- wmo/providers/openai.py +143 -0
- wmo/providers/openai_responses.py +240 -0
- wmo/providers/pool.py +170 -0
- wmo/providers/registry.py +73 -0
- wmo/providers/retry.py +151 -0
- wmo/providers/tinker.py +936 -0
- wmo/providers/waterfall.py +336 -0
- wmo/research/__init__.py +81 -0
- wmo/research/ablation.py +133 -0
- wmo/research/concurrency_plot.py +523 -0
- wmo/research/concurrency_run.py +240 -0
- wmo/research/concurrency_scaling.py +270 -0
- wmo/research/gepa_scaling.py +274 -0
- wmo/research/pipeline.py +198 -0
- wmo/research/scaling_split.py +82 -0
- wmo/research/scenario_fidelity.py +198 -0
- wmo/research/scenario_recovery.py +92 -0
- wmo/research/seed_stability.py +90 -0
- wmo/research/trace_scaling.py +348 -0
- wmo/retrieval/__init__.py +6 -0
- wmo/retrieval/embedders.py +105 -0
- wmo/retrieval/leakfree.py +52 -0
- wmo/retrieval/retriever.py +173 -0
- wmo/scenarios/__init__.py +58 -0
- wmo/scenarios/builder.py +152 -0
- wmo/scenarios/mining/__init__.py +27 -0
- wmo/scenarios/mining/clustering.py +171 -0
- wmo/scenarios/mining/facets.py +226 -0
- wmo/scenarios/mining/selection.py +220 -0
- wmo/scenarios/synthesis/__init__.py +6 -0
- wmo/scenarios/synthesis/scenario_set.py +63 -0
- wmo/scenarios/synthesis/synthesizer.py +85 -0
- wmo/scenarios/verification/__init__.py +17 -0
- wmo/scenarios/verification/judge.py +97 -0
- wmo/scenarios/verification/verify.py +135 -0
- wmo/serving/__init__.py +5 -0
- wmo/serving/builds.py +451 -0
- wmo/serving/chat.py +878 -0
- wmo/serving/endpoint_config.py +64 -0
- wmo/serving/savings.py +250 -0
- wmo/serving/server.py +553 -0
- wmo/serving/traces_source.py +206 -0
- wmo/telemetry.py +213 -0
- wmo/tracking/__init__.py +36 -0
- wmo/tracking/clock.py +24 -0
- wmo/tracking/metered.py +125 -0
- wmo/tracking/pricing.py +99 -0
- wmo/tracking/store.py +31 -0
- wmo/tracking/tracker.py +149 -0
- world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
- world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
- world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
- world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
wmo/tracking/pricing.py
ADDED
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
"""Per-model token pricing → USD cost.
|
|
2
|
+
|
|
3
|
+
Provider-agnostic: prices are keyed by a normalized model id (provider prefixes like Bedrock's
|
|
4
|
+
`us.anthropic.` are stripped before lookup), so the same Opus 4.8 row covers the direct API and
|
|
5
|
+
Bedrock. Prices are USD per 1M tokens; an unknown model costs 0.0 and is flagged so callers can
|
|
6
|
+
surface "cost unavailable" rather than silently under-reporting.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import re
|
|
12
|
+
|
|
13
|
+
from pydantic import BaseModel
|
|
14
|
+
|
|
15
|
+
from wmo.providers.base import TokenUsage
|
|
16
|
+
|
|
17
|
+
# Bedrock appends a snapshot date and/or version to the model id, e.g.
|
|
18
|
+
# `claude-haiku-4-5-20251001-v1:0` or `claude-opus-4-6-v1`. Strip them so the lookup key matches the
|
|
19
|
+
# undated table rows (`claude-haiku-4-5`). Only applied to `claude-*` ids.
|
|
20
|
+
_BEDROCK_SUFFIX = re.compile(r"(-\d{8})?(-v\d+)?(:\d+)?$")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class ModelPrice(BaseModel):
|
|
24
|
+
"""USD per 1,000,000 tokens, split by input/output."""
|
|
25
|
+
|
|
26
|
+
input_per_mtok: float
|
|
27
|
+
output_per_mtok: float
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
# Keyed by normalized model id (see `_normalize`). USD per 1M tokens.
|
|
31
|
+
#
|
|
32
|
+
# Completion prices verified 2026-06-25 against the live vendor pricing pages:
|
|
33
|
+
# - Claude: platform.claude.com/docs/en/about-claude/models/overview
|
|
34
|
+
# - OpenAI GPT-5.x: developers.openai.com/api/docs/pricing (Standard tier, short context)
|
|
35
|
+
# Embedding prices are long-stable list prices NOT re-fetched in that pass (the OpenAI pricing
|
|
36
|
+
# page no longer surfaces them); treat as approximate and re-verify if embed cost matters.
|
|
37
|
+
_PRICES: dict[str, ModelPrice] = {
|
|
38
|
+
# --- Anthropic / Bedrock (Claude) ---
|
|
39
|
+
"claude-fable-5": ModelPrice(input_per_mtok=10.0, output_per_mtok=50.0),
|
|
40
|
+
"claude-mythos-5": ModelPrice(input_per_mtok=10.0, output_per_mtok=50.0),
|
|
41
|
+
"claude-opus-4-8": ModelPrice(input_per_mtok=5.0, output_per_mtok=25.0),
|
|
42
|
+
"claude-opus-4-7": ModelPrice(input_per_mtok=5.0, output_per_mtok=25.0),
|
|
43
|
+
"claude-opus-4-6": ModelPrice(input_per_mtok=5.0, output_per_mtok=25.0),
|
|
44
|
+
"claude-opus-4-5": ModelPrice(input_per_mtok=5.0, output_per_mtok=25.0),
|
|
45
|
+
"claude-opus-4-1": ModelPrice(input_per_mtok=15.0, output_per_mtok=75.0),
|
|
46
|
+
"claude-sonnet-5": ModelPrice(input_per_mtok=3.0, output_per_mtok=15.0),
|
|
47
|
+
"claude-sonnet-4-6": ModelPrice(input_per_mtok=3.0, output_per_mtok=15.0),
|
|
48
|
+
"claude-haiku-4-5": ModelPrice(input_per_mtok=1.0, output_per_mtok=5.0),
|
|
49
|
+
# --- OpenAI / Azure OpenAI (GPT-5.x; Azure deployments reuse the base model's price) ---
|
|
50
|
+
"gpt-5.5": ModelPrice(input_per_mtok=5.0, output_per_mtok=30.0),
|
|
51
|
+
"gpt-5.5-pro": ModelPrice(input_per_mtok=30.0, output_per_mtok=180.0),
|
|
52
|
+
"gpt-5.4": ModelPrice(input_per_mtok=2.5, output_per_mtok=15.0),
|
|
53
|
+
"gpt-5.4-mini": ModelPrice(input_per_mtok=0.75, output_per_mtok=4.5),
|
|
54
|
+
"gpt-5.4-nano": ModelPrice(input_per_mtok=0.2, output_per_mtok=1.25),
|
|
55
|
+
# Self-hosted models (vLLM on our own GPUs) intentionally have NO row: their cost is amortized
|
|
56
|
+
# GPU time, not a per-token API price. `price_for` returns None for them, which the eval grid
|
|
57
|
+
# renders as "no cost"; a 0.0 ModelPrice would instead report a misleading $0.00.
|
|
58
|
+
# --- Embeddings (output tokens are always 0 for embed calls) ---
|
|
59
|
+
"text-embedding-3-small": ModelPrice(input_per_mtok=0.02, output_per_mtok=0.0),
|
|
60
|
+
"text-embedding-3-large": ModelPrice(input_per_mtok=0.13, output_per_mtok=0.0),
|
|
61
|
+
"amazon.titan-embed-text-v2:0": ModelPrice(input_per_mtok=0.02, output_per_mtok=0.0),
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _normalize(model: str) -> str:
|
|
66
|
+
"""Strip provider/region routing prefixes so one row covers a model across providers.
|
|
67
|
+
|
|
68
|
+
Bedrock ids look like `us.anthropic.claude-opus-4-8`; the direct API uses `claude-opus-4-8`.
|
|
69
|
+
We drop a leading region segment (`us.`/`eu.`/...) and an `anthropic.` vendor segment, but keep
|
|
70
|
+
`amazon.titan-...` (its `amazon.` is part of the canonical model id, not a routing prefix).
|
|
71
|
+
"""
|
|
72
|
+
normalized = model.strip()
|
|
73
|
+
region_prefixes = ("us.", "eu.", "apac.", "us-gov.", "global.", "jp.", "au.", "ca.")
|
|
74
|
+
for prefix in region_prefixes:
|
|
75
|
+
if normalized.startswith(prefix):
|
|
76
|
+
normalized = normalized[len(prefix) :]
|
|
77
|
+
break
|
|
78
|
+
if normalized.startswith("anthropic."):
|
|
79
|
+
normalized = normalized[len("anthropic.") :]
|
|
80
|
+
if normalized.startswith("claude-"):
|
|
81
|
+
# Drop a trailing Bedrock snapshot date / version (`-20251001-v1:0`, `-v1`) so dated
|
|
82
|
+
# inference-profile ids match the undated table rows.
|
|
83
|
+
normalized = _BEDROCK_SUFFIX.sub("", normalized)
|
|
84
|
+
return normalized
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def price_for(model: str) -> ModelPrice | None:
|
|
88
|
+
"""Return the price row for `model` (after normalization), or None if unknown."""
|
|
89
|
+
return _PRICES.get(_normalize(model))
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def cost_usd(model: str, usage: TokenUsage) -> float:
|
|
93
|
+
"""USD cost of `usage` on `model`. Unknown models cost 0.0 (see `price_for` to detect that)."""
|
|
94
|
+
price = price_for(model)
|
|
95
|
+
if price is None:
|
|
96
|
+
return 0.0
|
|
97
|
+
return (
|
|
98
|
+
usage.input_tokens * price.input_per_mtok + usage.output_tokens * price.output_per_mtok
|
|
99
|
+
) / 1_000_000
|
wmo/tracking/store.py
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""Persist + list run records under `.wmo/runs/`.
|
|
2
|
+
|
|
3
|
+
One JSON file per run (`<run_id>.json`). Kept tiny and dependency-free so `wmo build`/`serve` can
|
|
4
|
+
write a record without pulling in the rest of the harness.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
from wmo.tracking.tracker import RunRecord
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def save_run(record: RunRecord, runs_dir: str | Path) -> Path:
|
|
15
|
+
"""Write `record` to `<runs_dir>/<run_id>.json`, creating the directory if needed."""
|
|
16
|
+
path = Path(runs_dir)
|
|
17
|
+
path.mkdir(parents=True, exist_ok=True)
|
|
18
|
+
out = path / f"{record.run_id}.json"
|
|
19
|
+
out.write_text(record.model_dump_json(indent=2), encoding="utf-8")
|
|
20
|
+
return out
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def load_runs(runs_dir: str | Path) -> list[RunRecord]:
|
|
24
|
+
"""Load all run records from `runs_dir` (empty list if the directory doesn't exist)."""
|
|
25
|
+
path = Path(runs_dir)
|
|
26
|
+
if not path.exists():
|
|
27
|
+
return []
|
|
28
|
+
return [
|
|
29
|
+
RunRecord.model_validate_json(p.read_text(encoding="utf-8"))
|
|
30
|
+
for p in sorted(path.glob("*.json"))
|
|
31
|
+
]
|
wmo/tracking/tracker.py
ADDED
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
"""Run tracking: aggregate tokens → cost + wall-clock across the harness lifecycle.
|
|
2
|
+
|
|
3
|
+
A `RunTracker` collects `UsageEvent`s (one per LLM call, tagged by phase) and rolls them up into
|
|
4
|
+
`UsageTotals` (tokens, USD, calls) plus a wall-clock duration measured off an injectable `Clock`.
|
|
5
|
+
`RunRecord` is the persisted artifact (`.wmo/runs/<run_id>.json`).
|
|
6
|
+
|
|
7
|
+
The tracker is provider-agnostic: it records `(model, TokenUsage)` and prices via
|
|
8
|
+
`wmo.tracking.pricing`. It's fed at the provider boundary by `MeteredProvider` (so GEPA, the judge,
|
|
9
|
+
and the world model are all captured without touching the optimizer), and directly by the world
|
|
10
|
+
model's serve `step`.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import threading
|
|
16
|
+
from collections import defaultdict
|
|
17
|
+
from collections.abc import Iterator
|
|
18
|
+
from contextlib import contextmanager
|
|
19
|
+
from enum import StrEnum
|
|
20
|
+
|
|
21
|
+
from pydantic import BaseModel, Field
|
|
22
|
+
|
|
23
|
+
from wmo.providers.base import TokenUsage
|
|
24
|
+
from wmo.tracking.clock import Clock, SystemClock
|
|
25
|
+
from wmo.tracking.pricing import cost_usd
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class Phase(StrEnum):
|
|
29
|
+
"""Lifecycle phase a usage event is attributed to."""
|
|
30
|
+
|
|
31
|
+
BUILD = "build" # world-model build (overall)
|
|
32
|
+
GEPA = "gepa" # GEPA rollouts + reflection during optimization
|
|
33
|
+
JUDGE = "judge" # LLM-judge scoring
|
|
34
|
+
SERVE = "serve" # live world-model step calls
|
|
35
|
+
EMBED = "embed" # embedding calls (phi)
|
|
36
|
+
OTHER = "other"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class UsageEvent(BaseModel):
|
|
40
|
+
"""One metered LLM call."""
|
|
41
|
+
|
|
42
|
+
phase: Phase
|
|
43
|
+
model: str
|
|
44
|
+
usage: TokenUsage = Field(default_factory=TokenUsage)
|
|
45
|
+
cost_usd: float = 0.0
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
class UsageTotals(BaseModel):
|
|
49
|
+
"""Rolled-up usage: tokens, cost, and call count."""
|
|
50
|
+
|
|
51
|
+
calls: int = 0
|
|
52
|
+
input_tokens: int = 0
|
|
53
|
+
output_tokens: int = 0
|
|
54
|
+
cost_usd: float = 0.0
|
|
55
|
+
|
|
56
|
+
@property
|
|
57
|
+
def total_tokens(self) -> int:
|
|
58
|
+
return self.input_tokens + self.output_tokens
|
|
59
|
+
|
|
60
|
+
def _add(self, event: UsageEvent) -> None:
|
|
61
|
+
self.calls += 1
|
|
62
|
+
self.input_tokens += event.usage.input_tokens
|
|
63
|
+
self.output_tokens += event.usage.output_tokens
|
|
64
|
+
self.cost_usd += event.cost_usd
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class RunRecord(BaseModel):
|
|
68
|
+
"""Persisted summary of one run (build or serve session)."""
|
|
69
|
+
|
|
70
|
+
run_id: str
|
|
71
|
+
kind: str # "build" | "serve" | ... (free-form label for the run)
|
|
72
|
+
duration_seconds: float = 0.0
|
|
73
|
+
total: UsageTotals = Field(default_factory=UsageTotals)
|
|
74
|
+
by_phase: dict[Phase, UsageTotals] = Field(default_factory=dict)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class RunTracker:
|
|
78
|
+
"""Accumulates usage events and the wall-clock of a run.
|
|
79
|
+
|
|
80
|
+
Duration is measured between `start()` and `stop()` off the injected `Clock`, so tests can pass
|
|
81
|
+
a `FakeClock` and assert exact seconds. `record` is the single entry point for a metered call.
|
|
82
|
+
"""
|
|
83
|
+
|
|
84
|
+
def __init__(self, run_id: str, kind: str, clock: Clock | None = None) -> None:
|
|
85
|
+
self._run_id = run_id
|
|
86
|
+
self._kind = kind
|
|
87
|
+
self._clock = clock or SystemClock()
|
|
88
|
+
self._events: list[UsageEvent] = []
|
|
89
|
+
self._lock = threading.Lock()
|
|
90
|
+
self._started_at: float | None = None
|
|
91
|
+
self._elapsed: float = 0.0
|
|
92
|
+
|
|
93
|
+
def start(self) -> None:
|
|
94
|
+
self._started_at = self._clock.monotonic()
|
|
95
|
+
|
|
96
|
+
def stop(self) -> None:
|
|
97
|
+
if self._started_at is not None:
|
|
98
|
+
self._elapsed = self._clock.monotonic() - self._started_at
|
|
99
|
+
self._started_at = None
|
|
100
|
+
|
|
101
|
+
@contextmanager
|
|
102
|
+
def timed(self) -> Iterator[RunTracker]:
|
|
103
|
+
"""Time a run: `with tracker.timed(): ...` brackets start()/stop() even on error."""
|
|
104
|
+
self.start()
|
|
105
|
+
try:
|
|
106
|
+
yield self
|
|
107
|
+
finally:
|
|
108
|
+
self.stop()
|
|
109
|
+
|
|
110
|
+
def record(self, phase: Phase, model: str, usage: TokenUsage) -> UsageEvent:
|
|
111
|
+
"""Record one metered LLM call, pricing it via the model pricing table.
|
|
112
|
+
|
|
113
|
+
Thread-safe: GEPA evaluates batches concurrently, so metered calls land in parallel.
|
|
114
|
+
"""
|
|
115
|
+
event = UsageEvent(phase=phase, model=model, usage=usage, cost_usd=cost_usd(model, usage))
|
|
116
|
+
with self._lock:
|
|
117
|
+
self._events.append(event)
|
|
118
|
+
return event
|
|
119
|
+
|
|
120
|
+
@property
|
|
121
|
+
def events(self) -> list[UsageEvent]:
|
|
122
|
+
return list(self._events)
|
|
123
|
+
|
|
124
|
+
def totals(self) -> UsageTotals:
|
|
125
|
+
total = UsageTotals()
|
|
126
|
+
for event in self._events:
|
|
127
|
+
total._add(event)
|
|
128
|
+
return total
|
|
129
|
+
|
|
130
|
+
def by_phase(self) -> dict[Phase, UsageTotals]:
|
|
131
|
+
buckets: dict[Phase, UsageTotals] = defaultdict(UsageTotals)
|
|
132
|
+
for event in self._events:
|
|
133
|
+
buckets[event.phase]._add(event)
|
|
134
|
+
return dict(buckets)
|
|
135
|
+
|
|
136
|
+
def duration_seconds(self) -> float:
|
|
137
|
+
"""Elapsed seconds; live (since start) if still running, else the frozen final span."""
|
|
138
|
+
if self._started_at is not None:
|
|
139
|
+
return self._clock.monotonic() - self._started_at
|
|
140
|
+
return self._elapsed
|
|
141
|
+
|
|
142
|
+
def record_summary(self) -> RunRecord:
|
|
143
|
+
return RunRecord(
|
|
144
|
+
run_id=self._run_id,
|
|
145
|
+
kind=self._kind,
|
|
146
|
+
duration_seconds=self.duration_seconds(),
|
|
147
|
+
total=self.totals(),
|
|
148
|
+
by_phase=self.by_phase(),
|
|
149
|
+
)
|
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: world-model-optimizer
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Run agents, build world models from traces, and optimize agent harnesses.
|
|
5
|
+
Requires-Python: >=3.12
|
|
6
|
+
Requires-Dist: anthropic>=0.39
|
|
7
|
+
Requires-Dist: boto3>=1.34
|
|
8
|
+
Requires-Dist: click>=8.2
|
|
9
|
+
Requires-Dist: environment-capture
|
|
10
|
+
Requires-Dist: fastapi>=0.128
|
|
11
|
+
Requires-Dist: gepa>=0.1.1
|
|
12
|
+
Requires-Dist: httpx>=0.27
|
|
13
|
+
Requires-Dist: numpy>=1.26
|
|
14
|
+
Requires-Dist: openai>=1.40
|
|
15
|
+
Requires-Dist: posthog>=7.0
|
|
16
|
+
Requires-Dist: pydantic>=2.6
|
|
17
|
+
Requires-Dist: python-multipart>=0.0.9
|
|
18
|
+
Requires-Dist: rich>=14.1
|
|
19
|
+
Requires-Dist: scikit-learn>=1.4
|
|
20
|
+
Requires-Dist: tomli-w>=1.0
|
|
21
|
+
Requires-Dist: typer>=0.16
|
|
22
|
+
Requires-Dist: uvicorn>=0.38
|
|
23
|
+
Provides-Extra: connectors
|
|
24
|
+
Requires-Dist: mcp>=1.27; extra == 'connectors'
|
|
25
|
+
Provides-Extra: dev
|
|
26
|
+
Requires-Dist: e2b==2.31.0; extra == 'dev'
|
|
27
|
+
Requires-Dist: harbor==0.20.0; extra == 'dev'
|
|
28
|
+
Requires-Dist: matplotlib>=3.8; extra == 'dev'
|
|
29
|
+
Requires-Dist: mcp>=1.27; extra == 'dev'
|
|
30
|
+
Requires-Dist: opentelemetry-proto>=1.24; extra == 'dev'
|
|
31
|
+
Requires-Dist: pandas>=2.0; extra == 'dev'
|
|
32
|
+
Requires-Dist: psycopg[binary]>=3.1; extra == 'dev'
|
|
33
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
34
|
+
Requires-Dist: ruff>=0.5; extra == 'dev'
|
|
35
|
+
Requires-Dist: seaborn>=0.13; extra == 'dev'
|
|
36
|
+
Requires-Dist: tinker-cookbook<0.5,>=0.4.3; extra == 'dev'
|
|
37
|
+
Requires-Dist: tinker<0.24,>=0.23; extra == 'dev'
|
|
38
|
+
Requires-Dist: ty>=0.0.1a1; extra == 'dev'
|
|
39
|
+
Requires-Dist: wandb>=0.17; extra == 'dev'
|
|
40
|
+
Provides-Extra: distill
|
|
41
|
+
Requires-Dist: tinker-cookbook<0.5,>=0.4.3; extra == 'distill'
|
|
42
|
+
Requires-Dist: tinker<0.24,>=0.23; extra == 'distill'
|
|
43
|
+
Requires-Dist: wandb>=0.17; extra == 'distill'
|
|
44
|
+
Provides-Extra: e2b
|
|
45
|
+
Requires-Dist: e2b==2.31.0; extra == 'e2b'
|
|
46
|
+
Provides-Extra: harbor
|
|
47
|
+
Requires-Dist: harbor==0.20.0; extra == 'harbor'
|
|
48
|
+
Provides-Extra: otel
|
|
49
|
+
Requires-Dist: opentelemetry-proto>=1.24; extra == 'otel'
|
|
50
|
+
Provides-Extra: postgres
|
|
51
|
+
Requires-Dist: psycopg[binary]>=3.1; extra == 'postgres'
|
|
52
|
+
Provides-Extra: viz
|
|
53
|
+
Requires-Dist: matplotlib>=3.8; extra == 'viz'
|
|
54
|
+
Requires-Dist: pandas>=2.0; extra == 'viz'
|
|
55
|
+
Requires-Dist: seaborn>=0.13; extra == 'viz'
|
|
56
|
+
Description-Content-Type: text/markdown
|
|
57
|
+
|
|
58
|
+
# World Model Optimizer
|
|
59
|
+
|
|
60
|
+
`wmo` is an open-source project for running and building continuously improving agents. It
|
|
61
|
+
includes a flexible agent runtime, a world model that simulates tool calls, and an optimizer that
|
|
62
|
+
builds task-specific harnesses for stronger performance at lower cost.
|
|
63
|
+
|
|
64
|
+

|
|
65
|
+
|
|
66
|
+
<p align="center">
|
|
67
|
+
🌐 <a href="https://platform.experientiallabs.ai">Platform</a> |
|
|
68
|
+
📚 <a href="https://github.com/experientiallabs/world-model-optimizer/tree/main/docs">Docs</a> |
|
|
69
|
+
<a href="https://discord.gg/QwjJpEyHd"><img src="https://cdn.simpleicons.org/discord/5865F2" alt="" width="16" height="16"> Discord</a>
|
|
70
|
+
</p>
|
|
71
|
+
|
|
72
|
+
## Getting started
|
|
73
|
+
|
|
74
|
+
### Local setup
|
|
75
|
+
|
|
76
|
+
Install WMO, choose the model provider for the built-in runtime agent, and start a local run:
|
|
77
|
+
|
|
78
|
+
```bash
|
|
79
|
+
pip install world-model-optimizer
|
|
80
|
+
wmo providers set
|
|
81
|
+
wmo run --task "Inspect this repository and explain it"
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
Build a named world model from collected traces:
|
|
85
|
+
|
|
86
|
+
```bash
|
|
87
|
+
wmo build --file traces.jsonl --name my-environment
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
Then optimize an agent harness against that model and a set of tasks:
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
wmo optimize harness my-agent my-environment --tasks tasks.jsonl
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
### Hosted platform
|
|
97
|
+
|
|
98
|
+
Create an account at [platform.experientiallabs.ai](https://platform.experientiallabs.ai), then
|
|
99
|
+
authenticate the CLI:
|
|
100
|
+
|
|
101
|
+
```bash
|
|
102
|
+
wmo login
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
Copy an agent ID from the platform and run its current champion harness:
|
|
106
|
+
|
|
107
|
+
```bash
|
|
108
|
+
wmo run <agent-id>
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
### E2B backend
|
|
112
|
+
|
|
113
|
+
Hosted agents already run in platform-managed E2B sandboxes. To evaluate a local optimization in
|
|
114
|
+
E2B, install the extra and provide an E2B key:
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
pip install "world-model-optimizer[e2b]"
|
|
118
|
+
export E2B_API_KEY=...
|
|
119
|
+
wmo optimize harness my-agent my-environment --tasks tasks.jsonl --backend e2b
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
## Use a world model as an API
|
|
123
|
+
|
|
124
|
+
```python
|
|
125
|
+
from wmo import Action, ActionKind
|
|
126
|
+
from wmo.config.store import WorldModelStore
|
|
127
|
+
from wmo.engine.loader import load_world_model
|
|
128
|
+
|
|
129
|
+
model_dir = WorldModelStore(".wmo").resolve("airline")
|
|
130
|
+
wm, _provider = load_world_model(model_dir)
|
|
131
|
+
|
|
132
|
+
session = wm.new_session(task="check out the cart")
|
|
133
|
+
obs = wm.step(session.id, Action(kind=ActionKind.TOOL_CALL, name="add_to_cart",
|
|
134
|
+
arguments={"sku": "A1"}))
|
|
135
|
+
print(obs.content)
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
Or over HTTP (same code path), namespaced by model name: `GET /world_models`, then `POST /world_models/{name}/sessions` and `POST /world_models/{name}/sessions/{id}/step`.
|
|
139
|
+
|
|
140
|
+
## Run after platform login
|
|
141
|
+
|
|
142
|
+
After `wmo login`, the same `wmo run` command can open a hosted world model or run an agent's
|
|
143
|
+
current champion harness in E2B. The platform manages model and sandbox credentials, so hosted
|
|
144
|
+
runs do not need local API keys.
|
|
145
|
+
|
|
146
|
+
```bash
|
|
147
|
+
wmo login
|
|
148
|
+
wmo run <world-model-or-agent-id>
|
|
149
|
+
wmo run <agent-id> -u . --task "fix the failing tests"
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
Workspace upload is opt-in with `-u`: WMO live-syncs changes and preserves concurrent local edits.
|
|
153
|
+
Long-running agents can detach, continue in the platform, and be messaged or reattached later.
|
|
154
|
+
|
|
155
|
+
```bash
|
|
156
|
+
wmo run <agent-id> -u . --detach
|
|
157
|
+
wmo run --send "Now run the full test suite"
|
|
158
|
+
wmo run --attach
|
|
159
|
+
wmo run --end
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
## Runtime agents and optimizers in E2B sandboxes
|
|
163
|
+
|
|
164
|
+
WMO can run the real [pi](https://github.com/earendil-works/pi) worker inside isolated
|
|
165
|
+
[E2B](https://e2b.dev) sandboxes while the world model supplies the environment. Optimization and
|
|
166
|
+
evaluation rollouts run in parallel, and model credentials stay outside the sandbox.
|
|
167
|
+
|
|
168
|
+
```bash
|
|
169
|
+
wmo optimize harness my-agent my-environment --tasks tasks.jsonl --backend e2b
|
|
170
|
+
wmo eval tasks.jsonl --mode closed-loop --harness my-agent --harness-backend e2b
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
The optimizer can change prompts, tools, policies, skills, and runtime code. Every candidate is
|
|
174
|
+
measured against the same simulated tasks, and only changes that pass the evaluation gates become
|
|
175
|
+
the new versioned champion harness.
|
|
176
|
+
|
|
177
|
+
## Development
|
|
178
|
+
|
|
179
|
+
Managed with [uv](https://docs.astral.sh/uv/); linting/formatting with [ruff](https://docs.astral.sh/ruff/); type checking with [ty](https://github.com/astral-sh/ty). Conventions live in [AGENTS.md](./AGENTS.md).
|
|
180
|
+
|
|
181
|
+
```bash
|
|
182
|
+
uv sync --extra dev # env + dev tools
|
|
183
|
+
uv run ruff check . # lint
|
|
184
|
+
uv run ruff format . # format
|
|
185
|
+
uv run ty check # type check
|
|
186
|
+
uv run pytest -q # tests
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
## Usage telemetry
|
|
190
|
+
|
|
191
|
+
`wmo` uses anonymous usage telemetry to track the volume of usage.
|
|
192
|
+
Telemetry is strictly metadata. It never includes prompts, traces, actions, observations, file paths,
|
|
193
|
+
model names, provider credentials, or raw user content.
|
|
194
|
+
|
|
195
|
+
Telemetry is enabled by default. To opt out for a project:
|
|
196
|
+
|
|
197
|
+
```bash
|
|
198
|
+
uv run wmo config telemetry disable
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
This writes `.wmo/settings.toml`. You can re-enable it with `uv run wmo config telemetry enable`,
|
|
202
|
+
check the current setting with `uv run wmo config telemetry status`, or disable it for a process
|
|
203
|
+
with `DO_NOT_TRACK=1` or `WMO_TELEMETRY=0`.
|