world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,99 @@
1
+ """Per-model token pricing → USD cost.
2
+
3
+ Provider-agnostic: prices are keyed by a normalized model id (provider prefixes like Bedrock's
4
+ `us.anthropic.` are stripped before lookup), so the same Opus 4.8 row covers the direct API and
5
+ Bedrock. Prices are USD per 1M tokens; an unknown model costs 0.0 and is flagged so callers can
6
+ surface "cost unavailable" rather than silently under-reporting.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import re
12
+
13
+ from pydantic import BaseModel
14
+
15
+ from wmo.providers.base import TokenUsage
16
+
17
+ # Bedrock appends a snapshot date and/or version to the model id, e.g.
18
+ # `claude-haiku-4-5-20251001-v1:0` or `claude-opus-4-6-v1`. Strip them so the lookup key matches the
19
+ # undated table rows (`claude-haiku-4-5`). Only applied to `claude-*` ids.
20
+ _BEDROCK_SUFFIX = re.compile(r"(-\d{8})?(-v\d+)?(:\d+)?$")
21
+
22
+
23
+ class ModelPrice(BaseModel):
24
+ """USD per 1,000,000 tokens, split by input/output."""
25
+
26
+ input_per_mtok: float
27
+ output_per_mtok: float
28
+
29
+
30
+ # Keyed by normalized model id (see `_normalize`). USD per 1M tokens.
31
+ #
32
+ # Completion prices verified 2026-06-25 against the live vendor pricing pages:
33
+ # - Claude: platform.claude.com/docs/en/about-claude/models/overview
34
+ # - OpenAI GPT-5.x: developers.openai.com/api/docs/pricing (Standard tier, short context)
35
+ # Embedding prices are long-stable list prices NOT re-fetched in that pass (the OpenAI pricing
36
+ # page no longer surfaces them); treat as approximate and re-verify if embed cost matters.
37
+ _PRICES: dict[str, ModelPrice] = {
38
+ # --- Anthropic / Bedrock (Claude) ---
39
+ "claude-fable-5": ModelPrice(input_per_mtok=10.0, output_per_mtok=50.0),
40
+ "claude-mythos-5": ModelPrice(input_per_mtok=10.0, output_per_mtok=50.0),
41
+ "claude-opus-4-8": ModelPrice(input_per_mtok=5.0, output_per_mtok=25.0),
42
+ "claude-opus-4-7": ModelPrice(input_per_mtok=5.0, output_per_mtok=25.0),
43
+ "claude-opus-4-6": ModelPrice(input_per_mtok=5.0, output_per_mtok=25.0),
44
+ "claude-opus-4-5": ModelPrice(input_per_mtok=5.0, output_per_mtok=25.0),
45
+ "claude-opus-4-1": ModelPrice(input_per_mtok=15.0, output_per_mtok=75.0),
46
+ "claude-sonnet-5": ModelPrice(input_per_mtok=3.0, output_per_mtok=15.0),
47
+ "claude-sonnet-4-6": ModelPrice(input_per_mtok=3.0, output_per_mtok=15.0),
48
+ "claude-haiku-4-5": ModelPrice(input_per_mtok=1.0, output_per_mtok=5.0),
49
+ # --- OpenAI / Azure OpenAI (GPT-5.x; Azure deployments reuse the base model's price) ---
50
+ "gpt-5.5": ModelPrice(input_per_mtok=5.0, output_per_mtok=30.0),
51
+ "gpt-5.5-pro": ModelPrice(input_per_mtok=30.0, output_per_mtok=180.0),
52
+ "gpt-5.4": ModelPrice(input_per_mtok=2.5, output_per_mtok=15.0),
53
+ "gpt-5.4-mini": ModelPrice(input_per_mtok=0.75, output_per_mtok=4.5),
54
+ "gpt-5.4-nano": ModelPrice(input_per_mtok=0.2, output_per_mtok=1.25),
55
+ # Self-hosted models (vLLM on our own GPUs) intentionally have NO row: their cost is amortized
56
+ # GPU time, not a per-token API price. `price_for` returns None for them, which the eval grid
57
+ # renders as "no cost"; a 0.0 ModelPrice would instead report a misleading $0.00.
58
+ # --- Embeddings (output tokens are always 0 for embed calls) ---
59
+ "text-embedding-3-small": ModelPrice(input_per_mtok=0.02, output_per_mtok=0.0),
60
+ "text-embedding-3-large": ModelPrice(input_per_mtok=0.13, output_per_mtok=0.0),
61
+ "amazon.titan-embed-text-v2:0": ModelPrice(input_per_mtok=0.02, output_per_mtok=0.0),
62
+ }
63
+
64
+
65
+ def _normalize(model: str) -> str:
66
+ """Strip provider/region routing prefixes so one row covers a model across providers.
67
+
68
+ Bedrock ids look like `us.anthropic.claude-opus-4-8`; the direct API uses `claude-opus-4-8`.
69
+ We drop a leading region segment (`us.`/`eu.`/...) and an `anthropic.` vendor segment, but keep
70
+ `amazon.titan-...` (its `amazon.` is part of the canonical model id, not a routing prefix).
71
+ """
72
+ normalized = model.strip()
73
+ region_prefixes = ("us.", "eu.", "apac.", "us-gov.", "global.", "jp.", "au.", "ca.")
74
+ for prefix in region_prefixes:
75
+ if normalized.startswith(prefix):
76
+ normalized = normalized[len(prefix) :]
77
+ break
78
+ if normalized.startswith("anthropic."):
79
+ normalized = normalized[len("anthropic.") :]
80
+ if normalized.startswith("claude-"):
81
+ # Drop a trailing Bedrock snapshot date / version (`-20251001-v1:0`, `-v1`) so dated
82
+ # inference-profile ids match the undated table rows.
83
+ normalized = _BEDROCK_SUFFIX.sub("", normalized)
84
+ return normalized
85
+
86
+
87
+ def price_for(model: str) -> ModelPrice | None:
88
+ """Return the price row for `model` (after normalization), or None if unknown."""
89
+ return _PRICES.get(_normalize(model))
90
+
91
+
92
+ def cost_usd(model: str, usage: TokenUsage) -> float:
93
+ """USD cost of `usage` on `model`. Unknown models cost 0.0 (see `price_for` to detect that)."""
94
+ price = price_for(model)
95
+ if price is None:
96
+ return 0.0
97
+ return (
98
+ usage.input_tokens * price.input_per_mtok + usage.output_tokens * price.output_per_mtok
99
+ ) / 1_000_000
wmo/tracking/store.py ADDED
@@ -0,0 +1,31 @@
1
+ """Persist + list run records under `.wmo/runs/`.
2
+
3
+ One JSON file per run (`<run_id>.json`). Kept tiny and dependency-free so `wmo build`/`serve` can
4
+ write a record without pulling in the rest of the harness.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from pathlib import Path
10
+
11
+ from wmo.tracking.tracker import RunRecord
12
+
13
+
14
+ def save_run(record: RunRecord, runs_dir: str | Path) -> Path:
15
+ """Write `record` to `<runs_dir>/<run_id>.json`, creating the directory if needed."""
16
+ path = Path(runs_dir)
17
+ path.mkdir(parents=True, exist_ok=True)
18
+ out = path / f"{record.run_id}.json"
19
+ out.write_text(record.model_dump_json(indent=2), encoding="utf-8")
20
+ return out
21
+
22
+
23
+ def load_runs(runs_dir: str | Path) -> list[RunRecord]:
24
+ """Load all run records from `runs_dir` (empty list if the directory doesn't exist)."""
25
+ path = Path(runs_dir)
26
+ if not path.exists():
27
+ return []
28
+ return [
29
+ RunRecord.model_validate_json(p.read_text(encoding="utf-8"))
30
+ for p in sorted(path.glob("*.json"))
31
+ ]
@@ -0,0 +1,149 @@
1
+ """Run tracking: aggregate tokens → cost + wall-clock across the harness lifecycle.
2
+
3
+ A `RunTracker` collects `UsageEvent`s (one per LLM call, tagged by phase) and rolls them up into
4
+ `UsageTotals` (tokens, USD, calls) plus a wall-clock duration measured off an injectable `Clock`.
5
+ `RunRecord` is the persisted artifact (`.wmo/runs/<run_id>.json`).
6
+
7
+ The tracker is provider-agnostic: it records `(model, TokenUsage)` and prices via
8
+ `wmo.tracking.pricing`. It's fed at the provider boundary by `MeteredProvider` (so GEPA, the judge,
9
+ and the world model are all captured without touching the optimizer), and directly by the world
10
+ model's serve `step`.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import threading
16
+ from collections import defaultdict
17
+ from collections.abc import Iterator
18
+ from contextlib import contextmanager
19
+ from enum import StrEnum
20
+
21
+ from pydantic import BaseModel, Field
22
+
23
+ from wmo.providers.base import TokenUsage
24
+ from wmo.tracking.clock import Clock, SystemClock
25
+ from wmo.tracking.pricing import cost_usd
26
+
27
+
28
+ class Phase(StrEnum):
29
+ """Lifecycle phase a usage event is attributed to."""
30
+
31
+ BUILD = "build" # world-model build (overall)
32
+ GEPA = "gepa" # GEPA rollouts + reflection during optimization
33
+ JUDGE = "judge" # LLM-judge scoring
34
+ SERVE = "serve" # live world-model step calls
35
+ EMBED = "embed" # embedding calls (phi)
36
+ OTHER = "other"
37
+
38
+
39
+ class UsageEvent(BaseModel):
40
+ """One metered LLM call."""
41
+
42
+ phase: Phase
43
+ model: str
44
+ usage: TokenUsage = Field(default_factory=TokenUsage)
45
+ cost_usd: float = 0.0
46
+
47
+
48
+ class UsageTotals(BaseModel):
49
+ """Rolled-up usage: tokens, cost, and call count."""
50
+
51
+ calls: int = 0
52
+ input_tokens: int = 0
53
+ output_tokens: int = 0
54
+ cost_usd: float = 0.0
55
+
56
+ @property
57
+ def total_tokens(self) -> int:
58
+ return self.input_tokens + self.output_tokens
59
+
60
+ def _add(self, event: UsageEvent) -> None:
61
+ self.calls += 1
62
+ self.input_tokens += event.usage.input_tokens
63
+ self.output_tokens += event.usage.output_tokens
64
+ self.cost_usd += event.cost_usd
65
+
66
+
67
+ class RunRecord(BaseModel):
68
+ """Persisted summary of one run (build or serve session)."""
69
+
70
+ run_id: str
71
+ kind: str # "build" | "serve" | ... (free-form label for the run)
72
+ duration_seconds: float = 0.0
73
+ total: UsageTotals = Field(default_factory=UsageTotals)
74
+ by_phase: dict[Phase, UsageTotals] = Field(default_factory=dict)
75
+
76
+
77
+ class RunTracker:
78
+ """Accumulates usage events and the wall-clock of a run.
79
+
80
+ Duration is measured between `start()` and `stop()` off the injected `Clock`, so tests can pass
81
+ a `FakeClock` and assert exact seconds. `record` is the single entry point for a metered call.
82
+ """
83
+
84
+ def __init__(self, run_id: str, kind: str, clock: Clock | None = None) -> None:
85
+ self._run_id = run_id
86
+ self._kind = kind
87
+ self._clock = clock or SystemClock()
88
+ self._events: list[UsageEvent] = []
89
+ self._lock = threading.Lock()
90
+ self._started_at: float | None = None
91
+ self._elapsed: float = 0.0
92
+
93
+ def start(self) -> None:
94
+ self._started_at = self._clock.monotonic()
95
+
96
+ def stop(self) -> None:
97
+ if self._started_at is not None:
98
+ self._elapsed = self._clock.monotonic() - self._started_at
99
+ self._started_at = None
100
+
101
+ @contextmanager
102
+ def timed(self) -> Iterator[RunTracker]:
103
+ """Time a run: `with tracker.timed(): ...` brackets start()/stop() even on error."""
104
+ self.start()
105
+ try:
106
+ yield self
107
+ finally:
108
+ self.stop()
109
+
110
+ def record(self, phase: Phase, model: str, usage: TokenUsage) -> UsageEvent:
111
+ """Record one metered LLM call, pricing it via the model pricing table.
112
+
113
+ Thread-safe: GEPA evaluates batches concurrently, so metered calls land in parallel.
114
+ """
115
+ event = UsageEvent(phase=phase, model=model, usage=usage, cost_usd=cost_usd(model, usage))
116
+ with self._lock:
117
+ self._events.append(event)
118
+ return event
119
+
120
+ @property
121
+ def events(self) -> list[UsageEvent]:
122
+ return list(self._events)
123
+
124
+ def totals(self) -> UsageTotals:
125
+ total = UsageTotals()
126
+ for event in self._events:
127
+ total._add(event)
128
+ return total
129
+
130
+ def by_phase(self) -> dict[Phase, UsageTotals]:
131
+ buckets: dict[Phase, UsageTotals] = defaultdict(UsageTotals)
132
+ for event in self._events:
133
+ buckets[event.phase]._add(event)
134
+ return dict(buckets)
135
+
136
+ def duration_seconds(self) -> float:
137
+ """Elapsed seconds; live (since start) if still running, else the frozen final span."""
138
+ if self._started_at is not None:
139
+ return self._clock.monotonic() - self._started_at
140
+ return self._elapsed
141
+
142
+ def record_summary(self) -> RunRecord:
143
+ return RunRecord(
144
+ run_id=self._run_id,
145
+ kind=self._kind,
146
+ duration_seconds=self.duration_seconds(),
147
+ total=self.totals(),
148
+ by_phase=self.by_phase(),
149
+ )
@@ -0,0 +1,203 @@
1
+ Metadata-Version: 2.4
2
+ Name: world-model-optimizer
3
+ Version: 0.2.0
4
+ Summary: Run agents, build world models from traces, and optimize agent harnesses.
5
+ Requires-Python: >=3.12
6
+ Requires-Dist: anthropic>=0.39
7
+ Requires-Dist: boto3>=1.34
8
+ Requires-Dist: click>=8.2
9
+ Requires-Dist: environment-capture
10
+ Requires-Dist: fastapi>=0.128
11
+ Requires-Dist: gepa>=0.1.1
12
+ Requires-Dist: httpx>=0.27
13
+ Requires-Dist: numpy>=1.26
14
+ Requires-Dist: openai>=1.40
15
+ Requires-Dist: posthog>=7.0
16
+ Requires-Dist: pydantic>=2.6
17
+ Requires-Dist: python-multipart>=0.0.9
18
+ Requires-Dist: rich>=14.1
19
+ Requires-Dist: scikit-learn>=1.4
20
+ Requires-Dist: tomli-w>=1.0
21
+ Requires-Dist: typer>=0.16
22
+ Requires-Dist: uvicorn>=0.38
23
+ Provides-Extra: connectors
24
+ Requires-Dist: mcp>=1.27; extra == 'connectors'
25
+ Provides-Extra: dev
26
+ Requires-Dist: e2b==2.31.0; extra == 'dev'
27
+ Requires-Dist: harbor==0.20.0; extra == 'dev'
28
+ Requires-Dist: matplotlib>=3.8; extra == 'dev'
29
+ Requires-Dist: mcp>=1.27; extra == 'dev'
30
+ Requires-Dist: opentelemetry-proto>=1.24; extra == 'dev'
31
+ Requires-Dist: pandas>=2.0; extra == 'dev'
32
+ Requires-Dist: psycopg[binary]>=3.1; extra == 'dev'
33
+ Requires-Dist: pytest>=8.0; extra == 'dev'
34
+ Requires-Dist: ruff>=0.5; extra == 'dev'
35
+ Requires-Dist: seaborn>=0.13; extra == 'dev'
36
+ Requires-Dist: tinker-cookbook<0.5,>=0.4.3; extra == 'dev'
37
+ Requires-Dist: tinker<0.24,>=0.23; extra == 'dev'
38
+ Requires-Dist: ty>=0.0.1a1; extra == 'dev'
39
+ Requires-Dist: wandb>=0.17; extra == 'dev'
40
+ Provides-Extra: distill
41
+ Requires-Dist: tinker-cookbook<0.5,>=0.4.3; extra == 'distill'
42
+ Requires-Dist: tinker<0.24,>=0.23; extra == 'distill'
43
+ Requires-Dist: wandb>=0.17; extra == 'distill'
44
+ Provides-Extra: e2b
45
+ Requires-Dist: e2b==2.31.0; extra == 'e2b'
46
+ Provides-Extra: harbor
47
+ Requires-Dist: harbor==0.20.0; extra == 'harbor'
48
+ Provides-Extra: otel
49
+ Requires-Dist: opentelemetry-proto>=1.24; extra == 'otel'
50
+ Provides-Extra: postgres
51
+ Requires-Dist: psycopg[binary]>=3.1; extra == 'postgres'
52
+ Provides-Extra: viz
53
+ Requires-Dist: matplotlib>=3.8; extra == 'viz'
54
+ Requires-Dist: pandas>=2.0; extra == 'viz'
55
+ Requires-Dist: seaborn>=0.13; extra == 'viz'
56
+ Description-Content-Type: text/markdown
57
+
58
+ # World Model Optimizer
59
+
60
+ `wmo` is an open-source project for running and building continuously improving agents. It
61
+ includes a flexible agent runtime, a world model that simulates tool calls, and an optimizer that
62
+ builds task-specific harnesses for stronger performance at lower cost.
63
+
64
+ ![World model, runtime agent, and optimizer connected in a continuous improvement loop](assets/world-model-agent-loop.svg)
65
+
66
+ <p align="center">
67
+ 🌐 <a href="https://platform.experientiallabs.ai">Platform</a> |
68
+ 📚 <a href="https://github.com/experientiallabs/world-model-optimizer/tree/main/docs">Docs</a> |
69
+ <a href="https://discord.gg/QwjJpEyHd"><img src="https://cdn.simpleicons.org/discord/5865F2" alt="" width="16" height="16"> Discord</a>
70
+ </p>
71
+
72
+ ## Getting started
73
+
74
+ ### Local setup
75
+
76
+ Install WMO, choose the model provider for the built-in runtime agent, and start a local run:
77
+
78
+ ```bash
79
+ pip install world-model-optimizer
80
+ wmo providers set
81
+ wmo run --task "Inspect this repository and explain it"
82
+ ```
83
+
84
+ Build a named world model from collected traces:
85
+
86
+ ```bash
87
+ wmo build --file traces.jsonl --name my-environment
88
+ ```
89
+
90
+ Then optimize an agent harness against that model and a set of tasks:
91
+
92
+ ```bash
93
+ wmo optimize harness my-agent my-environment --tasks tasks.jsonl
94
+ ```
95
+
96
+ ### Hosted platform
97
+
98
+ Create an account at [platform.experientiallabs.ai](https://platform.experientiallabs.ai), then
99
+ authenticate the CLI:
100
+
101
+ ```bash
102
+ wmo login
103
+ ```
104
+
105
+ Copy an agent ID from the platform and run its current champion harness:
106
+
107
+ ```bash
108
+ wmo run <agent-id>
109
+ ```
110
+
111
+ ### E2B backend
112
+
113
+ Hosted agents already run in platform-managed E2B sandboxes. To evaluate a local optimization in
114
+ E2B, install the extra and provide an E2B key:
115
+
116
+ ```bash
117
+ pip install "world-model-optimizer[e2b]"
118
+ export E2B_API_KEY=...
119
+ wmo optimize harness my-agent my-environment --tasks tasks.jsonl --backend e2b
120
+ ```
121
+
122
+ ## Use a world model as an API
123
+
124
+ ```python
125
+ from wmo import Action, ActionKind
126
+ from wmo.config.store import WorldModelStore
127
+ from wmo.engine.loader import load_world_model
128
+
129
+ model_dir = WorldModelStore(".wmo").resolve("airline")
130
+ wm, _provider = load_world_model(model_dir)
131
+
132
+ session = wm.new_session(task="check out the cart")
133
+ obs = wm.step(session.id, Action(kind=ActionKind.TOOL_CALL, name="add_to_cart",
134
+ arguments={"sku": "A1"}))
135
+ print(obs.content)
136
+ ```
137
+
138
+ Or over HTTP (same code path), namespaced by model name: `GET /world_models`, then `POST /world_models/{name}/sessions` and `POST /world_models/{name}/sessions/{id}/step`.
139
+
140
+ ## Run after platform login
141
+
142
+ After `wmo login`, the same `wmo run` command can open a hosted world model or run an agent's
143
+ current champion harness in E2B. The platform manages model and sandbox credentials, so hosted
144
+ runs do not need local API keys.
145
+
146
+ ```bash
147
+ wmo login
148
+ wmo run <world-model-or-agent-id>
149
+ wmo run <agent-id> -u . --task "fix the failing tests"
150
+ ```
151
+
152
+ Workspace upload is opt-in with `-u`: WMO live-syncs changes and preserves concurrent local edits.
153
+ Long-running agents can detach, continue in the platform, and be messaged or reattached later.
154
+
155
+ ```bash
156
+ wmo run <agent-id> -u . --detach
157
+ wmo run --send "Now run the full test suite"
158
+ wmo run --attach
159
+ wmo run --end
160
+ ```
161
+
162
+ ## Runtime agents and optimizers in E2B sandboxes
163
+
164
+ WMO can run the real [pi](https://github.com/earendil-works/pi) worker inside isolated
165
+ [E2B](https://e2b.dev) sandboxes while the world model supplies the environment. Optimization and
166
+ evaluation rollouts run in parallel, and model credentials stay outside the sandbox.
167
+
168
+ ```bash
169
+ wmo optimize harness my-agent my-environment --tasks tasks.jsonl --backend e2b
170
+ wmo eval tasks.jsonl --mode closed-loop --harness my-agent --harness-backend e2b
171
+ ```
172
+
173
+ The optimizer can change prompts, tools, policies, skills, and runtime code. Every candidate is
174
+ measured against the same simulated tasks, and only changes that pass the evaluation gates become
175
+ the new versioned champion harness.
176
+
177
+ ## Development
178
+
179
+ Managed with [uv](https://docs.astral.sh/uv/); linting/formatting with [ruff](https://docs.astral.sh/ruff/); type checking with [ty](https://github.com/astral-sh/ty). Conventions live in [AGENTS.md](./AGENTS.md).
180
+
181
+ ```bash
182
+ uv sync --extra dev # env + dev tools
183
+ uv run ruff check . # lint
184
+ uv run ruff format . # format
185
+ uv run ty check # type check
186
+ uv run pytest -q # tests
187
+ ```
188
+
189
+ ## Usage telemetry
190
+
191
+ `wmo` uses anonymous usage telemetry to track the volume of usage.
192
+ Telemetry is strictly metadata. It never includes prompts, traces, actions, observations, file paths,
193
+ model names, provider credentials, or raw user content.
194
+
195
+ Telemetry is enabled by default. To opt out for a project:
196
+
197
+ ```bash
198
+ uv run wmo config telemetry disable
199
+ ```
200
+
201
+ This writes `.wmo/settings.toml`. You can re-enable it with `uv run wmo config telemetry enable`,
202
+ check the current setting with `uv run wmo config telemetry status`, or disable it for a process
203
+ with `DO_NOT_TRACK=1` or `WMO_TELEMETRY=0`.