world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,367 @@
1
+ """Max-fidelity auto-configuration: find the agentic config that best fits THIS corpus.
2
+
3
+ The lever matrix is empirical and task-dependent (measured across tau/terminal/swe: reasoning
4
+ wins on tool-call APIs, live fetch on web-heavy shells, the verify self-check on hard content
5
+ prediction — and no blanket setting wins everywhere). `wmo build --max-fidelity` automates that
6
+ search: each candidate configuration is replay-scored on the build's held-out split (leak-free,
7
+ same judge and demos as `wmo eval`), the winner's flags are persisted to the artifact's
8
+ config.toml, and serving picks them up automatically. The default build stays plain RAG — the
9
+ search is strictly opt-in, and `--fidelity-budget` chooses how deep it goes.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from collections.abc import Callable, Sequence
15
+ from dataclasses import dataclass
16
+ from statistics import fmean
17
+
18
+ from pydantic import BaseModel, Field
19
+
20
+ from wmo.core.types import Trace
21
+ from wmo.engine.grounding import FetchGrounder, Grounder, SourceResolver, extract_get_url
22
+ from wmo.engine.knowledge import seeded_knowledge_text
23
+ from wmo.engine.replay import replay
24
+ from wmo.engine.workspace import RepoTreeResolver
25
+ from wmo.optimize.judge import Judge
26
+ from wmo.providers.base import Embedder, Provider
27
+ from wmo.retrieval import EmbeddingRetriever
28
+
29
+ # Held-out traces scored per candidate by default: small enough that the search costs a fraction
30
+ # of the GEPA build, large enough to separate candidates beyond judge noise on most corpora.
31
+ DEFAULT_VAL_CAP = 8
32
+
33
+ # A challenger must beat the incumbent's mean fidelity by more than this to displace it. Sized
34
+ # above the per-cell judge/selection noise measured in the D37 ladder (±0.003–0.014): without
35
+ # it, a fluke +0.002 on the selection sample promoted a config that then LOST on test, which is
36
+ # exactly how the old independent-per-tier search went non-monotonic (tau high 0.886 < medium
37
+ # 0.891). The incumbent floor makes each tier improve-or-hold, never regress.
38
+ _NOISE_MARGIN = 0.01
39
+
40
+
41
+ @dataclass(frozen=True)
42
+ class CandidateConfig:
43
+ """One agentic configuration the search can select (maps 1:1 onto HarnessConfig flags)."""
44
+
45
+ label: str
46
+ reasoning: bool = False
47
+ knowledge: bool = False
48
+ verify: bool = False
49
+ grounder: str = "none"
50
+ # Workspace grounding (pinned source files + repo tree). Research-measurable today; enters
51
+ # build's search only once serve-side activation lands (a winner the runtime can't serve
52
+ # would be a lie in auto_fidelity.json).
53
+ workspace: bool = False
54
+ # Retrieval-depth overrides (None = the engine defaults: top_k 5, no demo cap). The measured
55
+ # "rag-deep" config: on record-heavy corpora more demos put more of the database in context
56
+ # (tau full-slice: base 0.939 -> 0.955 at k=20+cap2000, +0.016, replicating PR #72's +0.015);
57
+ # the cap keeps verbose corpora affordable.
58
+ top_k: int | None = None
59
+ demo_obs_cap: int | None = None
60
+
61
+
62
+ # Ordered by MEASURED serve cost, cheapest first (swe $/run: base 5.82, reason 6.20, workspace
63
+ # 6.48, kb 11.50, verify 12.73) — ties go to the earlier candidate, so the price-performance
64
+ # frontier wins: a cheap grounding config beats an expensive deliberation config that only
65
+ # matches it. Grounding-class candidates join right after `reason` because test-time ground
66
+ # truth is nearly free and measured as the largest lift class (fetch +0.040, workspace +0.065).
67
+ DEFAULT_CANDIDATES: tuple[CandidateConfig, ...] = (
68
+ CandidateConfig(label="base"),
69
+ CandidateConfig(label="reason", reasoning=True),
70
+ CandidateConfig(label="reason+kb", reasoning=True, knowledge=True),
71
+ CandidateConfig(label="reason+verify", reasoning=True, verify=True),
72
+ )
73
+ # Grounding-class candidates (cheap, corpus-gated). workspace needs instance pins (auto-detected
74
+ # next to the traces file); fetch is non-hermetic (hits the real web during the search) and is
75
+ # considered only when the corpus actually contains fetchable curl GETs.
76
+ WORKSPACE_CANDIDATE = CandidateConfig(label="reason+workspace", reasoning=True, workspace=True)
77
+ FETCH_CANDIDATE = CandidateConfig(label="reason+fetch", reasoning=True, grounder="fetch")
78
+ # Deep retrieval: 4x the demos, each observation capped (PR #72's optimized RAG, replicated on
79
+ # this protocol: tau +0.016 full-slice, terminal +0.004, swe +0.001 per #72). ~2x serve cost —
80
+ # it sits in the expensive tail, not the cheap frontier.
81
+ RAG_DEEP_CANDIDATE = CandidateConfig(label="rag-deep", top_k=20, demo_obs_cap=2000)
82
+ # The ladder's expensive tail: levers that ~2x the serve bill (kb rebuilds context, verify
83
+ # doubles completions). The medium tier's cheap-frontier search stops before these.
84
+ _EXPENSIVE_LABELS = frozenset({"rag-deep", "reason+kb", "reason+verify"})
85
+
86
+
87
+ @dataclass(frozen=True)
88
+ class CorpusSignature:
89
+ """Zero-token corpus features that predict which levers can pay off.
90
+
91
+ Measured reference points (healthy corpora, 2026-07-02): tau-bench curl=0.00/obs=414/
92
+ tool=1.00 (winner: reason), terminal-tasks 0.43/1236/0.00 (winner: reason+fetch),
93
+ swe-bench 0.00/889/0.00 (winner: reason+verify).
94
+ """
95
+
96
+ curl_get_share: float # steps whose action is a read-only curl GET
97
+ mean_obs_chars: float # content-heaviness of observations
98
+ tool_call_share: float # structured tool-call API vs free-form bash
99
+
100
+ @classmethod
101
+ def from_traces(cls, traces: list[Trace]) -> CorpusSignature:
102
+ steps = [s for t in traces for s in t.steps]
103
+ if not steps:
104
+ return cls(curl_get_share=0.0, mean_obs_chars=0.0, tool_call_share=0.0)
105
+ return cls(
106
+ curl_get_share=fmean(1.0 if extract_get_url(s.action) else 0.0 for s in steps),
107
+ mean_obs_chars=fmean(len(s.observation.content) for s in steps),
108
+ tool_call_share=fmean(0.0 if s.action.name == "bash" else 1.0 for s in steps),
109
+ )
110
+
111
+
112
+ def signature_estimate(signature: CorpusSignature, *, has_pins: bool = False) -> CandidateConfig:
113
+ """The single strongest config the measured lever matrix predicts for this corpus — free.
114
+
115
+ This is the `low` tier's shipped config and the incumbent FLOOR every searching tier seeds
116
+ from, so the ladder starts from a strong prior and can only improve. Deterministic from the
117
+ signature, so separate `wmo build --fidelity {medium,high,max}` invocations all carry the
118
+ identical floor. Rules are the D27 recommendations:
119
+ - tool-call APIs (tau-like): reasoning alone won (.899 -> .919).
120
+ - curl-heavy shells (terminal-like): live fetch of the action's own URL won (+0.040).
121
+ - pinned code repos (swe-like): workspace grounding won (+0.065).
122
+ - free-form, content-heavy bash otherwise: the knowledge base helped most.
123
+ - anything else: reasoning is the safe, cheap default.
124
+ """
125
+ if signature.tool_call_share >= 0.5:
126
+ return DEFAULT_CANDIDATES[1] # reason
127
+ if signature.curl_get_share >= 0.10:
128
+ return FETCH_CANDIDATE
129
+ if has_pins:
130
+ return WORKSPACE_CANDIDATE
131
+ if signature.mean_obs_chars >= 600:
132
+ return DEFAULT_CANDIDATES[2] # reason+kb
133
+ return DEFAULT_CANDIDATES[1] # reason
134
+
135
+
136
+ def select_candidates(
137
+ signature: CorpusSignature,
138
+ *,
139
+ full_ladder: bool = False,
140
+ has_pins: bool = False,
141
+ cheap_only: bool = False,
142
+ ) -> tuple[CandidateConfig, ...]:
143
+ """Choose which candidates are worth spending tokens on for THIS corpus.
144
+
145
+ Price sets the ORDER, never the menu: candidates are laddered cheapest-first (so a
146
+ truncated budget spends on cheap tricks first, and the winner tie-break favors the cheaper
147
+ config), but a candidate is dropped only when the corpus signature says it CANNOT matter
148
+ here — never because a cheaper lever is also available. Fidelity picks the winner; the
149
+ tier (cheap search vs `full_ladder`) only decides how hard we look.
150
+
151
+ Signature gates (from the measured lever matrix, not intuition):
152
+ - knowledge/verify: free-form (bash-like) environments; verify additionally wants
153
+ content-heavy observations (it only ever paid off where content prediction is hardest).
154
+ - fetch: a meaningful share of read-only curl GETs (nothing to prefetch = byte-identical
155
+ to `reason`); workspace: instance pins exist (same no-op logic).
156
+ `full_ladder` (the max tier) keeps only the no-op gates. `cheap_only` (the medium tier)
157
+ truncates the ladder before the expensive deliberation levers — grounding serves at ~base
158
+ cost, so even a budget tier can afford to discover a workspace/fetch win.
159
+ """
160
+ bash_like = signature.tool_call_share < 0.5
161
+ fetchable = signature.curl_get_share >= 0.10
162
+ # PRICE ORDER: base -> reason -> grounding class (workspace/fetch, ~free) -> kb -> verify.
163
+ if full_ladder:
164
+ chosen = [DEFAULT_CANDIDATES[0], DEFAULT_CANDIDATES[1]]
165
+ if has_pins:
166
+ chosen.append(WORKSPACE_CANDIDATE)
167
+ if fetchable:
168
+ chosen.append(FETCH_CANDIDATE)
169
+ chosen.append(RAG_DEEP_CANDIDATE)
170
+ chosen.extend([DEFAULT_CANDIDATES[2], DEFAULT_CANDIDATES[3]])
171
+ return _maybe_cheap(tuple(chosen), cheap_only)
172
+ chosen = [DEFAULT_CANDIDATES[0], DEFAULT_CANDIDATES[1]] # base, reason
173
+ if has_pins:
174
+ chosen.append(WORKSPACE_CANDIDATE) # cheapest strong lever when a repo pin exists
175
+ if fetchable:
176
+ chosen.append(FETCH_CANDIDATE)
177
+ chosen.append(RAG_DEEP_CANDIDATE) # never signature-gated: it never hurt anywhere measured
178
+ if bash_like:
179
+ chosen.append(DEFAULT_CANDIDATES[2]) # reason+kb
180
+ if signature.mean_obs_chars >= 600:
181
+ chosen.append(DEFAULT_CANDIDATES[3]) # reason+verify
182
+ return _maybe_cheap(tuple(chosen), cheap_only)
183
+
184
+
185
+ def _maybe_cheap(
186
+ candidates: tuple[CandidateConfig, ...], cheap_only: bool
187
+ ) -> tuple[CandidateConfig, ...]:
188
+ if not cheap_only:
189
+ return candidates
190
+ return tuple(c for c in candidates if c.label not in _EXPENSIVE_LABELS)
191
+
192
+
193
+ class WinnerSpec(BaseModel):
194
+ """The winning candidate's resolved flags, persisted so old artifacts stay self-describing.
195
+
196
+ Without this, `winner` is a foreign key into the in-code candidate tuple — and the ladder
197
+ churns (this PR alone added three candidates), so a rename would break `--max-fidelity`
198
+ loads of every previously built artifact.
199
+ """
200
+
201
+ label: str
202
+ reasoning: bool = False
203
+ knowledge: bool = False
204
+ verify: bool = False
205
+ grounder: str = "none"
206
+ workspace: bool = False
207
+ top_k: int | None = None
208
+ demo_obs_cap: int | None = None
209
+
210
+ @classmethod
211
+ def from_candidate(cls, candidate: CandidateConfig) -> WinnerSpec:
212
+ return cls(
213
+ label=candidate.label,
214
+ reasoning=candidate.reasoning,
215
+ knowledge=candidate.knowledge,
216
+ verify=candidate.verify,
217
+ grounder=candidate.grounder,
218
+ workspace=candidate.workspace,
219
+ top_k=candidate.top_k,
220
+ demo_obs_cap=candidate.demo_obs_cap,
221
+ )
222
+
223
+ def to_candidate(self) -> CandidateConfig:
224
+ return CandidateConfig(
225
+ label=self.label,
226
+ reasoning=self.reasoning,
227
+ knowledge=self.knowledge,
228
+ verify=self.verify,
229
+ grounder=self.grounder,
230
+ workspace=self.workspace,
231
+ top_k=self.top_k,
232
+ demo_obs_cap=self.demo_obs_cap,
233
+ )
234
+
235
+
236
+ class AutoFidelityReport(BaseModel):
237
+ """The search's outcome, persisted into the artifact for provenance."""
238
+
239
+ winner_label: str
240
+ scores: dict[str, float] = Field(default_factory=dict)
241
+ val_traces: int = 0
242
+ considered: list[str] = Field(default_factory=list) # candidate labels after pruning
243
+ # The winner's resolved flags (None only in pre-WinnerSpec artifacts, which fall back to
244
+ # the in-code label lookup).
245
+ winner_spec: WinnerSpec | None = None
246
+
247
+ @property
248
+ def winner(self) -> CandidateConfig:
249
+ if self.winner_spec is not None:
250
+ return self.winner_spec.to_candidate()
251
+ for candidate in (
252
+ *DEFAULT_CANDIDATES,
253
+ FETCH_CANDIDATE,
254
+ WORKSPACE_CANDIDATE,
255
+ RAG_DEEP_CANDIDATE,
256
+ ):
257
+ if candidate.label == self.winner_label:
258
+ return candidate
259
+ raise ValueError(f"unknown winner label {self.winner_label!r}")
260
+
261
+
262
+ def search_max_fidelity(
263
+ prompt: str,
264
+ train: list[Trace],
265
+ val: list[Trace],
266
+ provider: Provider,
267
+ judge: Judge,
268
+ embedder: Embedder | None,
269
+ *,
270
+ val_cap: int = DEFAULT_VAL_CAP,
271
+ top_k: int = 5,
272
+ seed: int = 0,
273
+ concurrency: int = 4,
274
+ candidates: Sequence[CandidateConfig] | None = None,
275
+ full_ladder: bool = False,
276
+ cheap_only: bool = False,
277
+ incumbent: CandidateConfig | None = None,
278
+ knowledge_text: str | None = None,
279
+ source_pins: str | None = None,
280
+ on_candidate_start: Callable[[str], None] | None = None,
281
+ on_candidate_done: Callable[[str, float], None] | None = None,
282
+ ) -> AutoFidelityReport:
283
+ """Replay-score the candidate configs on (a cap of) the held-out split; return the winner.
284
+
285
+ `candidates=None` computes the corpus signature (zero tokens) and prunes the ladder to the
286
+ levers that can matter for this corpus (`full_ladder=True` skips the pruning — the max
287
+ tier's "be certain" mode). Leak-free by construction: demos and the candidate knowledge
288
+ base both come from `train` only, and the scored `val` traces are the build's held-out
289
+ split.
290
+
291
+ `incumbent` is the floor the search may improve on but never regress below (the lower tier's
292
+ winner / the `low`-tier signature estimate). It is always scored on the same sample, and a
293
+ challenger only displaces it when it beats it by more than `_NOISE_MARGIN` — this is what
294
+ makes the tier ladder monotonic (improve-or-hold) instead of chasing selection noise. With
295
+ no incumbent the winner is just the highest mean fidelity, cheapest-config tie-break.
296
+ """
297
+ scored_val = val[:val_cap]
298
+ if candidates is None:
299
+ signature = CorpusSignature.from_traces(train)
300
+ candidates = select_candidates(
301
+ signature,
302
+ full_ladder=full_ladder,
303
+ has_pins=source_pins is not None,
304
+ cheap_only=cheap_only,
305
+ )
306
+ if incumbent is not None and not any(c.label == incumbent.label for c in candidates):
307
+ candidates = (incumbent, *candidates)
308
+ source = SourceResolver.from_file(source_pins) if source_pins is not None else None
309
+ tree = RepoTreeResolver(source.pins) if source is not None else None
310
+ # The candidate KB, seeded once (train-only) and reused for every knowledge candidate.
311
+ # `knowledge_text` lets the caller supply the EXACT text the artifact will serve (build
312
+ # seeds into the artifact dir first), so the winning score was measured on the KB that
313
+ # ships — a second independent extraction would be a different nondeterministic text.
314
+ kb_text = knowledge_text
315
+ if kb_text is None and any(c.knowledge for c in candidates):
316
+ kb_text = seeded_knowledge_text(train, provider)
317
+
318
+ scores: dict[str, float] = {}
319
+ best: CandidateConfig = incumbent if incumbent is not None else candidates[0]
320
+ best_score = -1.0
321
+ for candidate in candidates:
322
+ if on_candidate_start is not None:
323
+ on_candidate_start(candidate.label)
324
+ grounder: Grounder | None = FetchGrounder() if candidate.grounder == "fetch" else None
325
+ use_ws = candidate.workspace and source is not None
326
+ report = replay(
327
+ prompt,
328
+ scored_val,
329
+ provider,
330
+ judge,
331
+ retriever=EmbeddingRetriever(embedder) if embedder is not None else None,
332
+ train=train if embedder is not None else None,
333
+ top_k=candidate.top_k if candidate.top_k is not None else top_k,
334
+ max_retrieved_observation_chars=candidate.demo_obs_cap,
335
+ sample_turns="sampled",
336
+ seed=seed,
337
+ concurrency=concurrency,
338
+ knowledge=kb_text if candidate.knowledge else None,
339
+ reasoning=candidate.reasoning,
340
+ verify=candidate.verify,
341
+ grounder=grounder,
342
+ source=source if use_ws else None,
343
+ source_annotate_stale=use_ws,
344
+ tree=tree if use_ws else None,
345
+ )
346
+ scores[candidate.label] = report.mean_score
347
+ if on_candidate_done is not None:
348
+ on_candidate_done(candidate.label, report.mean_score)
349
+ if report.mean_score > best_score:
350
+ best, best_score = candidate, report.mean_score
351
+
352
+ if incumbent is not None:
353
+ # Improve-or-hold: keep the incumbent unless a challenger clears it by > the noise band.
354
+ incumbent_score = scores.get(incumbent.label, -1.0)
355
+ if best.label != incumbent.label and best_score <= incumbent_score + _NOISE_MARGIN:
356
+ # Defensive default: the incumbent is prepended into `candidates` above, so the
357
+ # lookup always finds it today — but a future caller passing an explicit
358
+ # `candidates` omitting it should fall back to the incumbent, not StopIteration.
359
+ best = next((c for c in candidates if c.label == incumbent.label), incumbent)
360
+
361
+ return AutoFidelityReport(
362
+ winner_label=best.label,
363
+ scores=scores,
364
+ val_traces=len(scored_val),
365
+ considered=[c.label for c in candidates],
366
+ winner_spec=WinnerSpec.from_candidate(best),
367
+ )