world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
wmo/config/config.py ADDED
@@ -0,0 +1,367 @@
1
+ """Project config + the `.wmo/` artifact layout.
2
+
3
+ `.wmo/` holds everything `wmo build` produces and `wmo serve` / `WorldModel.load` consume.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import tomllib
9
+ from dataclasses import dataclass
10
+ from enum import StrEnum
11
+ from pathlib import Path
12
+
13
+ import tomli_w
14
+ from pydantic import BaseModel, Field, JsonValue, ValidationError
15
+
16
+ from wmo.core.types import JsonObject
17
+ from wmo.providers.base import EmbedderKind, ProviderConfig, ProviderKind
18
+ from wmo.providers.models import resolve_provider_model
19
+
20
+ ARTIFACT_DIR = ".wmo"
21
+
22
+
23
+ class FidelityTier(StrEnum):
24
+ """Build-effort tiers: how much is spent making the artifact faithful.
25
+
26
+ The build command exposes ONLY this knob (raw iteration counts live in the Python API):
27
+ - low: RAG only — index the traces, ship the base prompt. Fast and near-free.
28
+ - medium: RAG + a light GEPA pass over the prompt.
29
+ - high: RAG + full GEPA + a cheap auto-config search (candidates pruned by corpus signature).
30
+ - max: RAG + deep GEPA + the full auto-config ladder, scored on more held-out traces.
31
+ """
32
+
33
+ LOW = "low"
34
+ MEDIUM = "medium"
35
+ HIGH = "high"
36
+ MAX = "max"
37
+
38
+
39
+ @dataclass(frozen=True)
40
+ class TierSpec:
41
+ """What one fidelity tier spends: GEPA rollouts, the auto-config search, retrieval phi."""
42
+
43
+ # GEPA optimization ITERATIONS (candidates proposed+evaluated). Each iteration costs about
44
+ # `minibatch + gepa_val_cap` metric calls (predict+judge pairs), so cost is bounded per tier
45
+ # regardless of corpus size — an uncapped valset once turned "budget 50" into ~7000 calls,
46
+ # and a TRACE-denominated cap let one long-trace corpus (swe: ~7 huge steps/trace) burn $131
47
+ # on a single "4-iteration" tier. STEPS are the unit that actually bounds cost.
48
+ gepa_budget: int
49
+ gepa_val_cap: int # STEPS GEPA selects candidates on (caps per-iteration cost)
50
+ config_search: bool
51
+ search_budget: int # held-out traces scored per candidate
52
+ full_ladder: bool # False = prune candidates by corpus signature
53
+ # True = search only the CHEAP frontier (base/reason/grounding class — levers that cost
54
+ # roughly nothing extra to serve). Tiers ration the expensive knobs (GEPA iterations,
55
+ # kb/verify scoring), but a nearly-free grounding win should be discoverable cheaply.
56
+ cheap_frontier_only: bool
57
+ # True = ship the corpus-signature's strongest ESTIMATED config with no LLM search (the
58
+ # `low` tier). It's the ladder's FLOOR: every searching tier seeds its incumbent from this
59
+ # same estimate and only replaces it on a clear win, so no searching tier ships worse than
60
+ # low. (This guarantees "tier >= low", NOT "high >= medium" — medium and high search
61
+ # different menus on different samples, so adjacent tiers are not strictly ordered; the
62
+ # floor is at low, not at the previous tier.)
63
+ estimate_only: bool
64
+ # Recommend provider-backed semantic phi. Kept as a field for explicit opt-in experiments,
65
+ # but NO tier sets it: semantic retrieval was measured WORSE than lexical hashing on every
66
+ # benchmark (PR #72 matrix: ada-002 terminal 0.790 vs hashing 0.818, swe 0.635 vs 0.640 —
67
+ # command outputs are predicted by literal token overlap, which char-trigrams capture and
68
+ # semantics blur), and the tier ladder's only semantic cells coincide with tau's decline
69
+ # (medium 0.891 hashing -> high 0.886 / max 0.882 semantic).
70
+ semantic_embeddings: bool
71
+
72
+
73
+ FIDELITY_TIERS: dict[FidelityTier, TierSpec] = {
74
+ # Low no longer means "plain base RAG": it ships the signature's strongest ESTIMATED
75
+ # config (reason / +fetch / +workspace / +kb by corpus shape — see `signature_estimate`)
76
+ # with zero LLM search. Free, and a strong floor the higher tiers can only build on.
77
+ FidelityTier.LOW: TierSpec(
78
+ gepa_budget=0,
79
+ gepa_val_cap=0,
80
+ config_search=False,
81
+ search_budget=0,
82
+ full_ladder=False,
83
+ cheap_frontier_only=False,
84
+ estimate_only=True,
85
+ semantic_embeddings=False,
86
+ ),
87
+ # Medium still searches the CHEAP frontier: grounding levers serve at ~base cost and score
88
+ # in a handful of traces, so even a budget tier can discover a workspace/fetch win. What
89
+ # medium rations is the expensive knobs — GEPA iterations and kb/verify scoring.
90
+ FidelityTier.MEDIUM: TierSpec(
91
+ gepa_budget=4,
92
+ gepa_val_cap=24,
93
+ config_search=True,
94
+ search_budget=4,
95
+ full_ladder=False,
96
+ cheap_frontier_only=True,
97
+ estimate_only=False,
98
+ semantic_embeddings=False,
99
+ ),
100
+ # High's GEPA stays at medium's 4 iterations: the 8-iteration increment measured ~noise
101
+ # on every benchmark once the config-search winners' known lifts are subtracted (ladder:
102
+ # tau -0.005, terminal +0.007, swe +0.006), and the GEPA scaling work (PR #97) found the
103
+ # base template's iteration lift ≈0 pre-fix. What high buys over medium: the full
104
+ # signature-pruned candidate menu (kb/verify) instead of the cheap frontier.
105
+ FidelityTier.HIGH: TierSpec(
106
+ gepa_budget=4,
107
+ gepa_val_cap=24,
108
+ config_search=True,
109
+ search_budget=8,
110
+ full_ladder=False,
111
+ cheap_frontier_only=False,
112
+ estimate_only=False,
113
+ semantic_embeddings=False,
114
+ ),
115
+ FidelityTier.MAX: TierSpec(
116
+ gepa_budget=16,
117
+ gepa_val_cap=32,
118
+ config_search=True,
119
+ search_budget=16,
120
+ full_ladder=True,
121
+ cheap_frontier_only=False,
122
+ estimate_only=False,
123
+ semantic_embeddings=False,
124
+ ),
125
+ }
126
+
127
+ # Env var names each provider backend reads its credentials from (documented for the user).
128
+ PROVIDER_ENV_VARS: dict[ProviderKind, list[str]] = {
129
+ ProviderKind.ANTHROPIC: ["ANTHROPIC_API_KEY"],
130
+ ProviderKind.BEDROCK: ["AWS_REGION", "AWS_ACCESS_KEY_ID", "AWS_SECRET_ACCESS_KEY"],
131
+ ProviderKind.AZURE_OPENAI: ["AZURE_OPENAI_API_KEY", "AZURE_OPENAI_ENDPOINT"],
132
+ ProviderKind.OPENAI: ["OPENAI_API_KEY"],
133
+ ProviderKind.OPENAI_RESPONSES: ["OPENAI_API_KEY"],
134
+ # Kept as a literal like every other entry (importing the provider module here would
135
+ # invert the config -> providers dependency); `wmo.providers.tinker.TINKER_API_KEY_ENV`
136
+ # is the name the provider actually reads, and a test pins the two together.
137
+ ProviderKind.TINKER: ["TINKER_API_KEY"],
138
+ }
139
+ """Every `ProviderKind` must appear here: the CLI's credential prompts, the picker's
140
+ "creds set" annotation, and the `providers verify` failure hint all read this dict, and a
141
+ missing kind silently degrades all three to "no credentials known"."""
142
+
143
+
144
+ class HarnessConfig(BaseModel):
145
+ """Persisted to `.wmo/config.toml` and reloaded by `wmo serve` / `WorldModel.load`."""
146
+
147
+ providers: list[ProviderConfig] = Field(default_factory=list)
148
+ serve_provider: ProviderKind = ProviderKind.ANTHROPIC # serves the live world model
149
+ # Which embedder supplies phi for retrieval. Defaults to the offline HashingEmbedder (no creds);
150
+ # set to a provider-backed kind (bedrock/openai/azure) for semantic phi.
151
+ embed_provider: EmbedderKind = EmbedderKind.HASHING
152
+ embed_dim: int = 512 # phi dimensionality; index + query embedder must agree on this
153
+ top_k: int = 5 # demos retrieved per step (DreamGym k)
154
+ # train/held-out ratio for GEPA; a proper fraction so both splits can be non-empty
155
+ train_split: float = Field(default=0.8, gt=0.0, lt=1.0)
156
+ gepa_budget: int = 10 # GEPA iterations; ~valset_cap calls each (see _cap_gepa_valset)
157
+ # Model id the GEPA judge runs on (same provider kind as serve). None = the serve model.
158
+ judge_model: str | None = None
159
+ trace_adapter: str = "otel-genai"
160
+ # Agentic-mode flags (all default OFF: artifacts built before these fields serve unchanged).
161
+ # `knowledge`: seed a knowledge base from train traces at build and render it into the env
162
+ # prompt at serve. `reasoning`: deliberate-then-answer output contract. `grounder`: web-search
163
+ # backend for grounding unknown entities ("none" keeps everything hermetic; see
164
+ # `wmo.engine.grounding` for backends).
165
+ knowledge: bool = False
166
+ reasoning: bool = False
167
+ grounder: str = "none"
168
+ # Second self-check completion per step (draft re-examined against the evidence). ~2x serve
169
+ # cost; earns it only where content prediction is hardest (empirically: swe-style suites).
170
+ verify: bool = False
171
+ # Verbalized confidence (WS-A6, D75): the contract asks for a 0.0-1.0 self-assessment of the
172
+ # emitted output (carried in Observation.metadata, never shown to the judge).
173
+ # `confidence_why` adds its one-line justification. Analysis/abstention lever, off by default.
174
+ confidence: bool = False
175
+ confidence_why: bool = False
176
+
177
+ def provider_config(self, kind: ProviderKind) -> ProviderConfig:
178
+ """Return the configured ProviderConfig for `kind` (model + backend knobs)."""
179
+ for pc in self.providers:
180
+ if pc.kind == kind:
181
+ return pc
182
+ raise ValueError(
183
+ f"no provider config for {kind.value}; configure it before building/serving "
184
+ f"(have: {[pc.kind.value for pc in self.providers]})"
185
+ )
186
+
187
+ def serve_provider_config(self) -> ProviderConfig:
188
+ """The ProviderConfig that serves the live world model."""
189
+ return self.provider_config(self.serve_provider)
190
+
191
+ def embed_provider_config(self) -> ProviderConfig:
192
+ """The ProviderConfig backing phi retrieval, with `embed_dim` stamped on.
193
+
194
+ Stamping `embed_dim` makes the backend request vectors of exactly the persisted dimension,
195
+ so the index and query embedders agree. Raises for `EmbedderKind.HASHING` (the offline
196
+ embedder has no provider) — guard with `embed_provider is EmbedderKind.HASHING` first.
197
+ """
198
+ config = self.provider_config(self.embed_provider.provider_kind())
199
+ return config.model_copy(update={"embed_dim": self.embed_dim})
200
+
201
+ @classmethod
202
+ def for_build(
203
+ cls,
204
+ *,
205
+ serve_provider: ProviderKind,
206
+ serve_model: str,
207
+ region: str | None,
208
+ embed_provider: EmbedderKind,
209
+ embed_model: str | None,
210
+ embed_dim: int,
211
+ gepa_budget: int,
212
+ train_split: float = 0.8,
213
+ judge_model: str | None = None,
214
+ trace_adapter: str = "otel-genai",
215
+ ) -> HarnessConfig:
216
+ """Assemble a build config from the choices `wmo build` collects.
217
+
218
+ Owns the one piece of provider wiring: a provider-backed embedder either **reuses** the
219
+ serve provider's config (same backend — just add `embed_model`) or gets **its own**
220
+ `ProviderConfig`. Keeping this here (not in the CLI) makes it unit-testable and gives every
221
+ entry point one place to construct a build config. Callers must already have validated that
222
+ a non-hashing embedder has an `embed_model`.
223
+ """
224
+ serve_spec = resolve_provider_model(serve_provider, serve_model)
225
+ serve = ProviderConfig(
226
+ kind=serve_provider,
227
+ model_type=serve_spec.model_type,
228
+ model=serve_spec.model_id,
229
+ region=region,
230
+ )
231
+ providers = [serve]
232
+ if embed_provider is not EmbedderKind.HASHING:
233
+ embed_kind = embed_provider.provider_kind()
234
+ if embed_kind == serve_provider:
235
+ providers[0] = serve.model_copy(update={"embed_model": embed_model})
236
+ else:
237
+ providers.append(
238
+ ProviderConfig(
239
+ kind=embed_kind,
240
+ model=embed_model or "",
241
+ embed_model=embed_model,
242
+ region=region,
243
+ )
244
+ )
245
+ return cls(
246
+ providers=providers,
247
+ serve_provider=serve_provider,
248
+ embed_provider=embed_provider,
249
+ embed_dim=embed_dim,
250
+ gepa_budget=gepa_budget,
251
+ train_split=train_split,
252
+ judge_model=(
253
+ resolve_provider_model(serve_provider, judge_model).model_id
254
+ if judge_model is not None
255
+ else None
256
+ ),
257
+ trace_adapter=trace_adapter,
258
+ )
259
+
260
+
261
+ class ArtifactPaths:
262
+ """Resolves the files under `.wmo/`."""
263
+
264
+ def __init__(self, root: str | Path = ARTIFACT_DIR) -> None:
265
+ self.root = Path(root)
266
+
267
+ @property
268
+ def config(self) -> Path:
269
+ return self.root / "config.toml"
270
+
271
+ @property
272
+ def traces(self) -> Path:
273
+ return self.root / "traces"
274
+
275
+ @property
276
+ def index(self) -> Path:
277
+ return self.root / "index"
278
+
279
+ @property
280
+ def runs(self) -> Path:
281
+ """Directory of persisted run records (build + serve), one JSON per run."""
282
+ return self.root / "runs"
283
+
284
+ @property
285
+ def base_prompt(self) -> Path:
286
+ return self.root / "prompts" / "base.txt"
287
+
288
+ @property
289
+ def optimized_prompt(self) -> Path:
290
+ return self.root / "prompts" / "optimized.txt"
291
+
292
+ @property
293
+ def frontier(self) -> Path:
294
+ return self.root / "prompts" / "frontier.json"
295
+
296
+ @property
297
+ def metrics(self) -> Path:
298
+ return self.root / "metrics.json"
299
+
300
+ @property
301
+ def knowledge(self) -> Path:
302
+ """Cross-session knowledge base directory (optional; absent on pre-knowledge artifacts)."""
303
+ return self.root / "knowledge"
304
+
305
+ @property
306
+ def auto_fidelity(self) -> Path:
307
+ """The auto-config search report (present on high/max-tier builds)."""
308
+ return self.root / "auto_fidelity.json"
309
+
310
+
311
+ def _strip_none(value: JsonValue) -> JsonValue:
312
+ """Drop `None`-valued keys recursively so TOML (which has no null) can represent the config.
313
+
314
+ On load, pydantic refills the missing optional fields with their `None` defaults, so dropping
315
+ them here round-trips losslessly.
316
+ """
317
+ if isinstance(value, dict):
318
+ return _strip_none_object(value)
319
+ if isinstance(value, list):
320
+ return [_strip_none(v) for v in value]
321
+ return value
322
+
323
+
324
+ def _strip_none_object(obj: dict[str, JsonValue]) -> JsonObject:
325
+ return {k: _strip_none(v) for k, v in obj.items() if v is not None}
326
+
327
+
328
+ def load_config(root: str | Path = ARTIFACT_DIR) -> HarnessConfig:
329
+ """Read `.wmo/config.toml`. Raises a friendly error if the project hasn't been built yet."""
330
+ paths = ArtifactPaths(root)
331
+ if not paths.root.exists():
332
+ raise FileNotFoundError(
333
+ f"no {ARTIFACT_DIR}/ directory at {paths.root}; run `wmo build` first to create it"
334
+ )
335
+ if not paths.config.exists():
336
+ raise FileNotFoundError(
337
+ f"{paths.config} is missing; run `wmo build` to (re)generate the project config"
338
+ )
339
+ try:
340
+ with paths.config.open("rb") as fh:
341
+ data = tomllib.load(fh)
342
+ except tomllib.TOMLDecodeError as exc:
343
+ raise ValueError(
344
+ f"{paths.config} is not valid TOML ({exc}); re-run `wmo build` to regenerate it"
345
+ ) from exc
346
+ try:
347
+ return HarnessConfig.model_validate(data)
348
+ except ValidationError as exc:
349
+ raise ValueError(
350
+ f"{paths.config} does not match the current config schema ({exc}); "
351
+ "re-run `wmo build` to regenerate it"
352
+ ) from exc
353
+
354
+
355
+ def save_config(config: HarnessConfig, root: str | Path = ARTIFACT_DIR) -> None:
356
+ """Write `config` to `.wmo/config.toml`, creating `.wmo/` if missing.
357
+
358
+ Writes to a temp file in the same directory and renames into place so an interrupted or
359
+ failed write never leaves a truncated `config.toml` behind.
360
+ """
361
+ paths = ArtifactPaths(root)
362
+ paths.root.mkdir(parents=True, exist_ok=True)
363
+ data = _strip_none_object(config.model_dump(mode="json"))
364
+ tmp = paths.config.with_name(f"{paths.config.name}.tmp")
365
+ with tmp.open("wb") as fh:
366
+ tomli_w.dump(data, fh)
367
+ tmp.replace(paths.config)
wmo/config/dotenv.py ADDED
@@ -0,0 +1,67 @@
1
+ """Minimal `.env` support: loaded on CLI startup, written by the wizard's credential prompts.
2
+
3
+ No third-party dotenv dependency — the harness only needs KEY=VALUE lines. Values entered in
4
+ the build wizard are persisted here so the next `wmo` invocation has them without re-prompting.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import os
10
+ import tempfile
11
+ from pathlib import Path
12
+
13
+ ENV_FILE = ".env"
14
+
15
+
16
+ def load_env_file(path: str | Path = ENV_FILE) -> None:
17
+ """Read KEY=VALUE lines from `path` into os.environ without overriding already-set vars."""
18
+ env_path = Path(path)
19
+ if not env_path.exists():
20
+ return
21
+ for raw_line in env_path.read_text(encoding="utf-8").splitlines():
22
+ line = raw_line.strip()
23
+ if not line or line.startswith("#") or "=" not in line:
24
+ continue
25
+ key, _, value = line.partition("=")
26
+ key, value = key.strip(), value.strip()
27
+ # Strip only a MATCHED surrounding quote pair; a secret legitimately ending in a
28
+ # quote character must survive the round-trip.
29
+ if len(value) >= 2 and value[0] == value[-1] and value[0] in "'\"":
30
+ value = value[1:-1]
31
+ if key and value and key not in os.environ:
32
+ os.environ[key] = value
33
+
34
+
35
+ def upsert_env_var(var: str, value: str, path: str | Path = ENV_FILE) -> None:
36
+ """Set `var` in os.environ and persist it to `path`, replacing any existing line for it.
37
+
38
+ Raises ValueError if `path` is a symlink: a credential rewrite must never end up in
39
+ whatever file the link happens to point at.
40
+ """
41
+ env_path = Path(path)
42
+ if env_path.is_symlink():
43
+ raise ValueError(
44
+ f"refusing to write credentials through the symlink {env_path}; "
45
+ f"set {var} in the link target or your shell instead"
46
+ )
47
+ os.environ[var] = value
48
+ lines = env_path.read_text(encoding="utf-8").splitlines() if env_path.exists() else []
49
+ rendered = f"{var}={value}"
50
+ for i, line in enumerate(lines):
51
+ if line.partition("=")[0].strip() == var:
52
+ lines[i] = rendered
53
+ break
54
+ else:
55
+ lines.append(rendered)
56
+ # Write-then-rename: mkstemp creates the temp file 0600 (owner-only, no umask window) and
57
+ # cannot hit a planted symlink; os.replace swaps the path atomically WITHOUT following a
58
+ # link that appeared after the check above, so the secret can never land in a linked-to
59
+ # file. This needs no platform-dependent open flags.
60
+ fd, tmp_name = tempfile.mkstemp(dir=env_path.parent, prefix=f"{env_path.name}.")
61
+ try:
62
+ with os.fdopen(fd, "w", encoding="utf-8") as fh:
63
+ fh.write("\n".join(lines) + "\n")
64
+ os.replace(tmp_name, env_path)
65
+ except BaseException:
66
+ os.unlink(tmp_name)
67
+ raise
wmo/config/settings.py ADDED
@@ -0,0 +1,128 @@
1
+ """Project-local settings stored under the selected harness root."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import tomllib
6
+ import uuid
7
+ from pathlib import Path
8
+
9
+ import tomli_w
10
+ from pydantic import BaseModel, Field, ValidationError
11
+
12
+ from wmo.config.config import ARTIFACT_DIR
13
+
14
+ SETTINGS_FILENAME = "settings.toml"
15
+
16
+
17
+ class TelemetrySettings(BaseModel):
18
+ """Usage telemetry preferences for this harness project."""
19
+
20
+ enabled: bool = True
21
+ anonymous_id: str | None = None
22
+
23
+
24
+ class ModelRole(BaseModel):
25
+ """One named model role: which provider/model handles this class of work."""
26
+
27
+ provider: str # a ProviderKind value ("bedrock", "azure", "openai", ...)
28
+ model: str
29
+ # Canonical model identity when `model` is a runtime id that does not carry it, e.g. a
30
+ # tinker:// weights path whose base model determines the renderer and tokenizer.
31
+ model_type: str | None = None
32
+ region: str | None = None # AWS Bedrock region
33
+ endpoint: str | None = None # Azure OpenAI / custom base URL
34
+ deployment: str | None = None # Azure OpenAI deployment name
35
+ api_version: str | None = None # Azure OpenAI API version (azure roles get a default)
36
+ reasoning_effort: str | None = None # structured reasoning effort, when supported
37
+
38
+
39
+ class ModelsSettings(BaseModel):
40
+ """Role-based model defaults for this project (`.wmo/settings.toml`, `[models.<role>]`).
41
+
42
+ Five roles keep the surface small: `worker` does quality-critical generation (scenario
43
+ synthesis, cluster naming, agent rollouts); `judge` grades (checklist judging, inline
44
+ validity gates) and should be a different model family from `worker` so the grader carries
45
+ no self-preference bias toward the generator's outputs; `summary` does high-volume cheap
46
+ extraction (trace facets/digests); `meta` is the harness-search delta proposer, which
47
+ needs a long-context, long-output model (one proposal holds every harness surface in its
48
+ prompt and replies with a complete replacement surface); `agent` is the agent-under-test
49
+ whose harness `wmo optimize harness` searches. Set it to optimize a harness for a model
50
+ distinct from the world model's serve provider (e.g. a small self-hosted agent against a
51
+ frontier-served world model). Unset `judge`/`summary` fall back to `worker`; each command
52
+ documents which explicit model flags and opt-in roles it uses.
53
+ """
54
+
55
+ worker: ModelRole | None = None
56
+ judge: ModelRole | None = None
57
+ summary: ModelRole | None = None
58
+ meta: ModelRole | None = None
59
+ agent: ModelRole | None = None
60
+
61
+ def resolve(self, role: str) -> ModelRole | None:
62
+ """The configured role, with unset `judge`/`summary` falling back to `worker`.
63
+
64
+ `meta` and `agent` deliberately do NOT fall back to `worker`: each is picked for a
65
+ need the scenario worker does not serve (the proposer's long-context/long-output
66
+ surface; the agent-under-test's own identity), so when unset they return None and the
67
+ caller keeps its own default (`wmo optimize harness` and closed-loop eval use the world
68
+ model's provider for the opt-in roles when unset).
69
+ """
70
+ if role not in ("worker", "judge", "summary", "meta", "agent"):
71
+ raise ValueError(
72
+ f"unknown model role {role!r}; expected worker, judge, summary, meta, or agent"
73
+ )
74
+ configured: ModelRole | None = getattr(self, role)
75
+ if role in ("meta", "agent"):
76
+ return configured
77
+ return configured or self.worker
78
+
79
+
80
+ class ProjectSettings(BaseModel):
81
+ """Settings that are local to one harness project root."""
82
+
83
+ telemetry: TelemetrySettings = Field(default_factory=TelemetrySettings)
84
+ models: ModelsSettings = Field(default_factory=ModelsSettings)
85
+
86
+
87
+ def settings_path(root: str | Path = ARTIFACT_DIR) -> Path:
88
+ return Path(root) / SETTINGS_FILENAME
89
+
90
+
91
+ def load_settings(root: str | Path = ARTIFACT_DIR) -> ProjectSettings:
92
+ path = settings_path(root)
93
+ if not path.exists():
94
+ return ProjectSettings()
95
+ try:
96
+ with path.open("rb") as fh:
97
+ data = tomllib.load(fh)
98
+ except tomllib.TOMLDecodeError as exc:
99
+ raise ValueError(f"{path} is not valid TOML ({exc})") from exc
100
+ try:
101
+ return ProjectSettings.model_validate(data)
102
+ except ValidationError as exc:
103
+ raise ValueError(f"{path} does not match the current settings schema ({exc})") from exc
104
+
105
+
106
+ def save_settings(settings: ProjectSettings, root: str | Path = ARTIFACT_DIR) -> None:
107
+ path = settings_path(root)
108
+ path.parent.mkdir(parents=True, exist_ok=True)
109
+ data = settings.model_dump(mode="json", exclude_none=True)
110
+ tmp = path.with_name(f"{path.name}.tmp")
111
+ with tmp.open("wb") as fh:
112
+ tomli_w.dump(data, fh)
113
+ tmp.replace(path)
114
+
115
+
116
+ def set_telemetry_enabled(enabled: bool, root: str | Path = ARTIFACT_DIR) -> ProjectSettings:
117
+ settings = load_settings(root)
118
+ settings.telemetry.enabled = enabled
119
+ save_settings(settings, root)
120
+ return settings
121
+
122
+
123
+ def ensure_telemetry_anonymous_id(root: str | Path = ARTIFACT_DIR) -> str:
124
+ settings = load_settings(root)
125
+ if settings.telemetry.anonymous_id is None:
126
+ settings.telemetry.anonymous_id = uuid.uuid4().hex
127
+ save_settings(settings, root)
128
+ return settings.telemetry.anonymous_id