world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,140 @@
1
+ """Exact task selection for the harbor scorer.
2
+
3
+ Harbor's own `DatasetConfig.task_names` filter uses fnmatch semantics, so a task id containing a
4
+ glob character would silently over-match, and an optimizer's train/heldout split firewall relies
5
+ on exact selection. This module resolves a dataset once, post-filters by exact id, downloads any
6
+ remote (git/package) tasks ONCE, and returns pinned `TaskConfig`s that candidate jobs run
7
+ directly (`tasks=[...]`, `overwrite=False`): per-candidate jobs must never re-clone or clobber
8
+ the shared task cache that concurrent jobs are reading from.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import re
14
+ from collections.abc import Sequence
15
+ from pathlib import Path
16
+ from typing import Protocol
17
+
18
+ from harbor.models.job.config import DatasetConfig
19
+ from harbor.models.trial.config import TaskConfig
20
+ from harbor.tasks.client import BatchDownloadResult, TaskClient, TaskIdType
21
+
22
+ _GIT_COMMIT_PATTERN = re.compile(r"^[0-9a-fA-F]{40}$|^[0-9a-fA-F]{64}$")
23
+
24
+
25
+ class HarborTaskDownloader(Protocol):
26
+ """The task-download slice of harbor's TaskClient (fakes replace it in tests)."""
27
+
28
+ async def download_tasks(
29
+ self,
30
+ task_ids: list[TaskIdType],
31
+ overwrite: bool = False,
32
+ output_dir: Path | None = None,
33
+ ) -> BatchDownloadResult: ...
34
+
35
+
36
+ async def resolve_harbor_tasks(
37
+ dataset: DatasetConfig | Path,
38
+ task_ids: Sequence[str],
39
+ *,
40
+ task_client: HarborTaskDownloader | None = None,
41
+ ) -> list[TaskConfig]:
42
+ """Resolve `task_ids` from a harbor dataset (or local task dir) by exact identity.
43
+
44
+ Args:
45
+ dataset: A harbor `DatasetConfig`, or a local directory of task dirs (shorthand for
46
+ `DatasetConfig(path=...)`).
47
+ task_ids: Exact task names to select, in the order the caller wants them evaluated.
48
+ task_client: Download seam; defaults to harbor's `TaskClient`.
49
+
50
+ Returns:
51
+ One pinned `TaskConfig` per requested id, in request order. Git and package tasks are
52
+ downloaded here exactly once (git with `overwrite=True`, since harbor's cache does not
53
+ verify that an existing checkout still matches the requested commit) and returned with
54
+ the resolved commit pinned and `overwrite=False`, so every candidate job reuses the
55
+ resolved local bytes instead of re-cloning or clobbering the shared cache.
56
+
57
+ Raises:
58
+ ValueError: On empty/duplicate ids, a dataset that resolves duplicate task names, ids
59
+ the dataset does not contain, or a git download whose commit cannot be pinned.
60
+ """
61
+ requested = list(task_ids)
62
+ if not requested or any(not task_id for task_id in requested):
63
+ raise ValueError("task_ids must be nonempty strings")
64
+ if len(requested) != len(set(requested)):
65
+ raise ValueError("task_ids must be unique")
66
+
67
+ if isinstance(dataset, Path):
68
+ dataset = DatasetConfig(path=dataset)
69
+ # Resolve the dataset WITHOUT harbor's filters, then select by exact id ourselves:
70
+ # task_names is an fnmatch pattern list, not an exact-selection API.
71
+ resolved = DatasetConfig.model_validate(dataset.model_dump(mode="python"))
72
+ resolved.task_names = None
73
+ resolved.exclude_task_names = None
74
+ resolved.n_tasks = None
75
+ configs = await resolved.get_task_configs()
76
+
77
+ by_id: dict[str, TaskConfig] = {}
78
+ for config in configs:
79
+ task_id = config.get_task_id().get_name()
80
+ if task_id in by_id:
81
+ raise ValueError(f"harbor dataset resolved duplicate task {task_id!r}")
82
+ by_id[task_id] = config
83
+ missing = sorted(set(requested) - set(by_id))
84
+ if missing:
85
+ raise ValueError(
86
+ f"harbor task selection was not exact: missing={missing}; "
87
+ "check the ids against the dataset's task names"
88
+ )
89
+
90
+ selected = [by_id[task_id].model_copy(deep=True) for task_id in requested]
91
+ return await _pin_remote_tasks(
92
+ selected,
93
+ dataset=resolved,
94
+ task_client=task_client or TaskClient(),
95
+ )
96
+
97
+
98
+ async def _pin_remote_tasks(
99
+ selected: list[TaskConfig],
100
+ *,
101
+ dataset: DatasetConfig,
102
+ task_client: HarborTaskDownloader,
103
+ ) -> list[TaskConfig]:
104
+ """Download remote tasks once and pin their provenance with `overwrite=False`."""
105
+ remote_indexes = [
106
+ index
107
+ for index, config in enumerate(selected)
108
+ if config.is_git_task() or config.is_package_task()
109
+ ]
110
+ if not remote_indexes:
111
+ return selected
112
+ downloads = await task_client.download_tasks(
113
+ [selected[index].get_task_id() for index in remote_indexes],
114
+ # Refresh git checkouts once, here: harbor's cache does not verify that an existing
115
+ # checkout still came from the requested commit.
116
+ overwrite=dataset.overwrite
117
+ or any(selected[index].is_git_task() for index in remote_indexes),
118
+ output_dir=dataset.download_dir,
119
+ )
120
+ if len(downloads.results) != len(remote_indexes):
121
+ raise ValueError("harbor returned an incomplete task download result")
122
+ for index, download in zip(remote_indexes, downloads.results, strict=True):
123
+ config = selected[index]
124
+ updates: dict[str, object] = {"overwrite": False}
125
+ if config.is_git_task():
126
+ commit = download.resolved_git_commit_id
127
+ if commit is None or _GIT_COMMIT_PATTERN.fullmatch(commit) is None:
128
+ raise ValueError(
129
+ "harbor did not resolve a git commit for task "
130
+ f"{config.get_task_id().get_name()!r}"
131
+ )
132
+ requested_commit = config.git_commit_id
133
+ if requested_commit is not None and requested_commit.lower() != commit.lower():
134
+ raise ValueError(
135
+ "harbor resolved a different git commit for task "
136
+ f"{config.get_task_id().get_name()!r}"
137
+ )
138
+ updates["git_commit_id"] = commit.lower()
139
+ selected[index] = config.model_copy(update=updates)
140
+ return selected
wmo/evals/open_loop.py ADDED
@@ -0,0 +1,194 @@
1
+ """Open-loop evaluation: reconstruction fidelity over trace files (the default `wmo eval` mode).
2
+
3
+ `replay` (in `wmo.engine.replay`) scores one corpus of held-out steps teacher-forced. This
4
+ orchestration layer is what `wmo eval` calls: it loads one or more OTel trace files, splits each
5
+ into train/holdout, replays the holdout through a world-model prompt with leak-free RAG, and
6
+ aggregates a per-file + overall scorecard. Its closed-loop counterpart
7
+ (`wmo eval --mode closed-loop`, `wmo.evals.closed_loop`) runs a live agent instead of replaying;
8
+ both implement the `Evaluation` interface in `wmo.evals.base`.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from pathlib import Path
14
+ from statistics import fmean, pstdev
15
+
16
+ from pydantic import BaseModel, Field
17
+
18
+ from wmo.engine.build import split_traces, split_traces_3way
19
+ from wmo.engine.knowledge import seeded_knowledge_text
20
+ from wmo.engine.replay import ReplayReport, replay, valid_scores
21
+ from wmo.ingest import get_adapter
22
+ from wmo.optimize.judge import Judge
23
+ from wmo.providers.base import Embedder, Provider
24
+ from wmo.retrieval import EmbeddingRetriever
25
+
26
+
27
+ class EvalReport(BaseModel):
28
+ """Per-file fidelity reports plus the step-weighted overall mean ± std.
29
+
30
+ `per_file` maps a trace file's clean name to its `ReplayReport` (per-step `StepResult`s), and
31
+ `overall_fidelity`/`overall_std` are the step-weighted aggregates across files.
32
+ """
33
+
34
+ per_file: dict[str, ReplayReport] = Field(default_factory=dict)
35
+ overall_fidelity: float = 0.0 # step-weighted mean of valid per-step scores across all files
36
+ overall_std: float = 0.0 # std of valid per-step scores across all files
37
+ total_steps: int = 0 # all steps attempted, including judge-invalid ones
38
+ total_invalid: int = 0 # judge failures across files; excluded from fidelity/std
39
+
40
+ @property
41
+ def headline(self) -> float:
42
+ """The `EvalResult` headline: per-step reconstruction fidelity."""
43
+ return self.overall_fidelity
44
+
45
+ @property
46
+ def total_valid(self) -> int:
47
+ """Steps that actually back the fidelity mean (judge-invalid ones excluded)."""
48
+ return self.total_steps - self.total_invalid
49
+
50
+ def summary(self) -> str:
51
+ invalid = f", {self.total_invalid} judge-invalid excluded" if self.total_invalid else ""
52
+ return (
53
+ f"fidelity={self.overall_fidelity:.3f}±{self.overall_std:.3f} "
54
+ f"({self.total_steps} steps, {len(self.per_file)} file(s){invalid})"
55
+ )
56
+
57
+
58
+ def evaluate_files(
59
+ files: list[Path],
60
+ prompt: str,
61
+ provider: Provider,
62
+ judge: Judge,
63
+ *,
64
+ embedder: Embedder | None = None,
65
+ train_split: float = 0.7,
66
+ val_frac: float | None = None,
67
+ top_k: int = 5,
68
+ sample_turns: str = "all",
69
+ seed: int = 0,
70
+ adapter_name: str = "otel-genai",
71
+ max_holdout_traces: int | None = None,
72
+ knowledge: bool = False,
73
+ reasoning: bool = False,
74
+ ) -> EvalReport:
75
+ """Replay-score each trace file's held-out split. `embedder=None` -> zero-shot (no retrieval).
76
+
77
+ Each file is split deterministically; tiny corpora with no held-out trace fall back to scoring
78
+ every trace. RAG, when enabled, retrieves from that file's own train split only (leak-free).
79
+ `sample_turns`/`seed` are forwarded to `replay` (see its docstring). `max_holdout_traces` caps
80
+ how many held-out traces are scored per file (a deterministic prefix by trace_id) - for cheap
81
+ dry-runs; the train side stays full so retrieval is unaffected.
82
+
83
+ `val_frac` makes the split leak-free against a GEPA-evolved prompt: when it is a positive
84
+ fraction, the traces are cut 3-way (`train`/`val`/`test`) on the same hash line GEPA used, and
85
+ only the reserved `test` band is scored, so a prompt selected on `val` is never graded on those
86
+ same traces. Retrieval still draws from `train` only. `val_frac` of `None` or `0` keeps the
87
+ plain 2-way `train`/held-out split (`split_traces_3way` requires a strictly positive band).
88
+
89
+ `knowledge` seeds an ephemeral knowledge base from each file's TRAIN split (never the holdout -
90
+ the same leak-free discipline as RAG) and renders it into every prediction; `reasoning`
91
+ switches predictions to the deliberate-then-answer contract. Both mirror the serving engine's
92
+ agentic mode. Closed-loop evals get agentic mode from the ARTIFACT instead (the served
93
+ WorldModel's config / --max-fidelity winner), not from these flags.
94
+ """
95
+ adapter = get_adapter(adapter_name)
96
+ per_file: dict[str, ReplayReport] = {}
97
+ for path in files:
98
+ traces = adapter.from_file(str(path))
99
+ if not traces:
100
+ continue
101
+ if val_frac: # truthy (a positive val band); None or 0.0 keeps the plain 2-way split
102
+ train, _val, holdout = split_traces_3way(traces, train_split, val_frac)
103
+ else:
104
+ train, holdout = split_traces(traces, train_split)
105
+ if not holdout: # tiny corpus: evaluate on everything
106
+ train, holdout = traces, traces
107
+ if max_holdout_traces is not None:
108
+ holdout = sorted(holdout, key=lambda t: t.trace_id)[:max_holdout_traces]
109
+ retriever = EmbeddingRetriever(embedder) if embedder is not None else None
110
+ # Ephemeral, per-file, train-only KB: rendered text only — nothing under models/ is read
111
+ # or written, so eval can never leak a serve-time learned.md into scoring.
112
+ knowledge_text = seeded_knowledge_text(train, provider) if knowledge else None
113
+ name = _display_name(path)
114
+ per_file[name] = replay(
115
+ prompt,
116
+ holdout,
117
+ provider,
118
+ judge,
119
+ retriever=retriever,
120
+ train=train if embedder is not None else None,
121
+ top_k=top_k,
122
+ sample_turns=sample_turns,
123
+ seed=seed,
124
+ knowledge=knowledge_text,
125
+ reasoning=reasoning,
126
+ )
127
+
128
+ # Step-weighted aggregate over every validly-judged step across files (judge failures are
129
+ # counted in total_invalid, never as spurious zeros — see replay.valid_scores).
130
+ step_scores = valid_scores(r for rep in per_file.values() for r in rep.results)
131
+ overall = fmean(step_scores) if step_scores else 0.0
132
+ overall_std = pstdev(step_scores) if len(step_scores) > 1 else 0.0
133
+ return EvalReport(
134
+ per_file=per_file,
135
+ overall_fidelity=overall,
136
+ overall_std=overall_std,
137
+ total_steps=sum(rep.n_steps for rep in per_file.values()),
138
+ total_invalid=sum(rep.n_invalid for rep in per_file.values()),
139
+ )
140
+
141
+
142
+ def _display_name(path: Path) -> str:
143
+ """Human label for a corpus, using the example folder name for `traces.otel.jsonl`."""
144
+ name = path.name.removesuffix(".jsonl").removesuffix(".otel")
145
+ return path.parent.name if name == "traces" else name
146
+
147
+
148
+ class OpenLoopEval:
149
+ """The open-loop `Evaluation`: teacher-forced replay of held-out trace steps."""
150
+
151
+ def __init__(
152
+ self,
153
+ files: list[Path],
154
+ prompt: str,
155
+ provider: Provider,
156
+ judge: Judge,
157
+ *,
158
+ embedder: Embedder | None = None,
159
+ train_split: float = 0.7,
160
+ top_k: int = 5,
161
+ sample_turns: str = "all",
162
+ seed: int = 0,
163
+ adapter_name: str = "otel-genai",
164
+ knowledge: bool = False,
165
+ reasoning: bool = False,
166
+ ) -> None:
167
+ self._files = files
168
+ self._prompt = prompt
169
+ self._provider = provider
170
+ self._judge = judge
171
+ self._embedder = embedder
172
+ self._train_split = train_split
173
+ self._top_k = top_k
174
+ self._sample_turns = sample_turns
175
+ self._seed = seed
176
+ self._adapter_name = adapter_name
177
+ self._knowledge = knowledge
178
+ self._reasoning = reasoning
179
+
180
+ def run(self) -> EvalReport:
181
+ return evaluate_files(
182
+ self._files,
183
+ self._prompt,
184
+ self._provider,
185
+ self._judge,
186
+ embedder=self._embedder,
187
+ train_split=self._train_split,
188
+ top_k=self._top_k,
189
+ sample_turns=self._sample_turns,
190
+ seed=self._seed,
191
+ adapter_name=self._adapter_name,
192
+ knowledge=self._knowledge,
193
+ reasoning=self._reasoning,
194
+ )
wmo/evals/tasks.py ADDED
@@ -0,0 +1,53 @@
1
+ """Task specs for closed-loop evaluation: an instruction plus gold assertions that define success.
2
+
3
+ Gold assertions are semantic post-conditions the `GoldJudge` checks against the run transcript —
4
+ conditions on the final state, made robust to wording by an LLM judge instead of brittle exact
5
+ matching. Tasks are typically derived from the same benchmark the world model's traces came from —
6
+ `Trace.metadata` already carries gold assertions for traces captured with them.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from pathlib import Path
12
+
13
+ from pydantic import BaseModel, Field, field_validator
14
+
15
+ from wmo.core.text import normalize_durable_text
16
+
17
+
18
+ class TaskSpec(BaseModel):
19
+ """One task: what the agent must do, and the assertions that must hold afterwards."""
20
+
21
+ task_id: str
22
+ instruction: str
23
+ gold: list[str] = Field(default_factory=list) # assertions that define success
24
+
25
+ @field_validator("task_id", "instruction")
26
+ @classmethod
27
+ def _normalize_scalar_text(cls, value: str) -> str:
28
+ return normalize_durable_text(value)
29
+
30
+ @field_validator("gold")
31
+ @classmethod
32
+ def _normalize_gold(cls, value: list[str]) -> list[str]:
33
+ return [normalize_durable_text(assertion) for assertion in value]
34
+
35
+
36
+ def load_tasks(path: str | Path) -> list[TaskSpec]:
37
+ """Read a JSONL task file (one TaskSpec per line; blank lines ignored).
38
+
39
+ Duplicate `task_id`s are an error: reports key outcomes by task_id, so a duplicate would run
40
+ (and cost) k passes twice while silently keeping only the last outcome.
41
+ """
42
+ tasks: list[TaskSpec] = []
43
+ for line in Path(path).read_text(encoding="utf-8").splitlines():
44
+ stripped = line.strip()
45
+ if stripped:
46
+ tasks.append(TaskSpec.model_validate_json(stripped))
47
+ if not tasks:
48
+ raise ValueError(f"no tasks in {path}")
49
+ ids = [t.task_id for t in tasks]
50
+ duplicates = sorted({i for i in ids if ids.count(i) > 1})
51
+ if duplicates:
52
+ raise ValueError(f"duplicate task_id(s) in {path}: {duplicates}")
53
+ return tasks
@@ -0,0 +1,51 @@
1
+ """The agent harness: the scaffold a live agent runs with, and the machinery to improve it.
2
+
3
+ A minimal, fixed agent loop (`AgentRuntime`) drives one action at a time against an
4
+ `AgentEnvironment` — an interface, not a backend: closed-loop eval (`wmo.evals.closed_loop`) binds
5
+ it to the world model, and a real execution backend can bind the same loop to reality. What the
6
+ loop runs with is a `HarnessDoc` — a typed document of identity-keyed surfaces (prompt sections,
7
+ tool policy, loop params, skills) stored as immutable versions with movable aliases
8
+ (`wmo.harness.store`) and updated through audited `HarnessDelta`s (`wmo.harness.delta`,
9
+ docs/reference/harness_delta.md) proposed by a meta-agent and gated on non-regression
10
+ (`wmo.harness.create`, the `wmo optimize` search).
11
+
12
+ `create` and `mutate` are imported directly (not re-exported here): they depend on
13
+ `wmo.evals.closed_loop`, which itself binds to this package's runtime — re-exporting them would
14
+ make `import wmo.evals` observe a partially initialized module.
15
+ """
16
+
17
+ from wmo.harness.delta import (
18
+ FailureSignature,
19
+ GateRecord,
20
+ HarnessDelta,
21
+ SurfaceOp,
22
+ apply_delta,
23
+ )
24
+ from wmo.harness.doc import HarnessDoc, Surface, SurfaceKind
25
+ from wmo.harness.environment import AgentEnvironment, is_env_action
26
+ from wmo.harness.runtime import AgentRuntime, RunResult, StopReason
27
+ from wmo.harness.skills import Skill, SkillLibrary
28
+ from wmo.harness.store import HarnessStore
29
+ from wmo.harness.tools import TOOL_REGISTRY, ToolCall, parse_tool_call
30
+
31
+ __all__ = [
32
+ "TOOL_REGISTRY",
33
+ "AgentEnvironment",
34
+ "AgentRuntime",
35
+ "FailureSignature",
36
+ "GateRecord",
37
+ "HarnessDelta",
38
+ "HarnessDoc",
39
+ "HarnessStore",
40
+ "RunResult",
41
+ "Skill",
42
+ "SkillLibrary",
43
+ "StopReason",
44
+ "Surface",
45
+ "SurfaceKind",
46
+ "SurfaceOp",
47
+ "ToolCall",
48
+ "apply_delta",
49
+ "is_env_action",
50
+ "parse_tool_call",
51
+ ]