world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
wmo/harness/doc.py ADDED
@@ -0,0 +1,556 @@
1
+ """`HarnessDoc`: a harness as a typed document of identity-keyed surfaces.
2
+
3
+ A harness is not a directory of files — it is a set of named **surfaces**, each an independently
4
+ addressable unit of behavior: prompt sections, the tool policy, scalar loop parameters, and skills.
5
+ Files (`SYSTEM.md`, `config.toml`, `skills/*.md`) are a *render target* the store exports for
6
+ running the harness elsewhere; the document is the interface everything else programs against.
7
+
8
+ Why surfaces instead of files:
9
+ - **Identity.** Every surface has a stable id (`prompt:core`, `skill:count-words`). An update names
10
+ its target; nothing is ever addressed by position or filename, so "which thing changed" is never
11
+ inferred.
12
+ - **Content addressing.** Each surface has a content hash, and the document has a hash over its
13
+ surfaces. "The score of harness X" is well-defined because X is a hash; an update can assert
14
+ exactly what it believes it is editing.
15
+ - **Typed validation.** A document validates as a whole (tools resolve, `submit` present, params in
16
+ range, budgets respected) the moment it is constructed — an invalid harness cannot exist as a
17
+ value, so nothing downstream re-checks.
18
+
19
+ Surface *content* stays a free-form string on purpose: structure lives in the envelope (ids, kinds,
20
+ hashes, budgets), not in the payload, so richer surface kinds can be added without changing how
21
+ updates work.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ import hashlib
27
+ import os
28
+ import re
29
+ from collections.abc import Callable
30
+ from enum import StrEnum
31
+ from pathlib import PurePosixPath
32
+ from typing import TYPE_CHECKING
33
+
34
+ from pydantic import BaseModel, Field, field_validator, model_validator
35
+
36
+ from wmo.core.text import validate_durable_text
37
+ from wmo.harness.code_runtime import (
38
+ DEFAULT_RUNTIME_CODE,
39
+ CodeRuntime,
40
+ compile_harness_code,
41
+ )
42
+ from wmo.harness.runtime import (
43
+ DEFAULT_EVAL_EPISODE_TIMEOUT_S,
44
+ DEFAULT_MAX_OUTPUT_TOKENS,
45
+ DEFAULT_MAX_TURNS,
46
+ DEFAULT_SYSTEM_PROMPT,
47
+ AgentRuntime,
48
+ Runtime,
49
+ strip_json_protocol_clause,
50
+ validate_episode_timeout_s,
51
+ )
52
+ from wmo.harness.skills import Skill, SkillLibrary
53
+ from wmo.harness.tools import DEFAULT_TOOLS, READ_SKILL, render_tools, resolve_tools
54
+ from wmo.providers.base import Provider, ToolCallingProvider
55
+
56
+ if TYPE_CHECKING:
57
+ # Import-time neutral: pi_e2b (the optional e2b extra's consumer) is imported lazily inside
58
+ # runtime(); this name exists only for the e2b_pool annotation.
59
+ from wmo.harness.pi_e2b import E2BSandboxPool
60
+
61
+ _SLUG_RE = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+)*$")
62
+
63
+ # Well-known surface ids. The tool policy and scalar parameters are singletons; prompt and skill
64
+ # surfaces may be added freely (an update can split `prompt:core` into finer sections).
65
+ TOOL_POLICY_ID = "tool_policy:main"
66
+ MAX_TURNS_ID = "param:max-turns"
67
+ MAX_OUTPUT_TOKENS_ID = "param:max-output-tokens"
68
+ TEMPERATURE_ID = "param:temperature"
69
+ RUNTIME_KIND_ID = "param:runtime-kind" # absent/"kit-python" -> in-process; "pi-node" -> PiRuntime
70
+ CODE_RUNTIME_ID = "code:runtime"
71
+
72
+ _SAFE_PATH_RE = re.compile(r"^[A-Za-z0-9._-]+(?:/[A-Za-z0-9._-]+)*$")
73
+
74
+ # Store metadata filenames a pathful code surface may never materialize to: on render they
75
+ # would shadow the harness store's own authority files.
76
+ _STORE_METADATA_FILES = frozenset({"doc.json", "aliases.toml"})
77
+ MAX_SURFACE_PATH_BYTES = 1_024
78
+
79
+ DEFAULT_TEMPERATURE = 0.7
80
+
81
+
82
+ def code_surface_id(relpath: str) -> str:
83
+ """A stable surface id from a path (`src/agent-loop.ts` -> `code:src-agent-loop-ts`)."""
84
+ return "code:" + relpath.replace("/", "-").replace(".", "-")
85
+
86
+
87
+ class SurfaceKind(StrEnum):
88
+ PROMPT = "prompt" # a section of the system prompt (joined in id order)
89
+ SKILL = "skill" # one skill: frontmatter (name, description) + body
90
+ TOOL_POLICY = "tool_policy" # the tool list, one tool name per line
91
+ PARAM = "param" # a scalar loop knob, serialized as its string form
92
+ CODE = "code" # the agent loop itself: a module defining `run(kit)` (see code_runtime)
93
+
94
+
95
+ class Surface(BaseModel):
96
+ """One named, independently addressable unit of harness behavior."""
97
+
98
+ id: str # "<kind>:<slug>"
99
+ kind: SurfaceKind
100
+ content: str
101
+ # For CODE surfaces of a vendored multi-file harness: the file path the content materializes
102
+ # to (relative, no traversal). A path-less CODE surface is the legacy in-process
103
+ # `code:runtime` module.
104
+ path: str | None = None
105
+ # Optional size budget (characters). Enforced at construction: a surface that exceeds its
106
+ # budget is invalid, so context cost is a schema property rather than a runtime surprise.
107
+ budget: int | None = Field(default=None, ge=1)
108
+
109
+ @model_validator(mode="after")
110
+ def _validate(self) -> Surface:
111
+ validate_durable_text(self.content, field=f"surface {self.id!r} content")
112
+ prefix, sep, slug = self.id.partition(":")
113
+ if not sep or prefix != self.kind.value or not _SLUG_RE.fullmatch(slug):
114
+ raise ValueError(
115
+ f"surface id {self.id!r} must be '{self.kind.value}:<kebab-slug>' matching its kind"
116
+ )
117
+ if self.path is not None:
118
+ if self.kind is not SurfaceKind.CODE:
119
+ raise ValueError(f"surface {self.id!r}: only code surfaces may carry a path")
120
+ candidate = PurePosixPath(self.path)
121
+ if (
122
+ not _SAFE_PATH_RE.fullmatch(self.path)
123
+ or not candidate.parts
124
+ or candidate.as_posix() != self.path
125
+ or ".." in candidate.parts
126
+ or len(self.path.encode("utf-8")) > MAX_SURFACE_PATH_BYTES
127
+ ):
128
+ raise ValueError(
129
+ f"surface {self.id!r}: unsafe path {self.path!r}; a code surface path must "
130
+ f"be a canonical relative POSIX path of at most {MAX_SURFACE_PATH_BYTES} "
131
+ "UTF-8 bytes with no '.' or '..' segments"
132
+ )
133
+ if self.path in _STORE_METADATA_FILES:
134
+ raise ValueError(
135
+ f"surface {self.id!r}: path {self.path!r} would shadow a harness store "
136
+ "metadata file; rename the file"
137
+ )
138
+ if self.id == CODE_RUNTIME_ID:
139
+ raise ValueError(
140
+ f"surface {CODE_RUNTIME_ID!r} is the in-process runtime module and must not "
141
+ f"carry a path (got {self.path!r}); rename the file so it maps to its own id"
142
+ )
143
+ expected_id = code_surface_id(self.path)
144
+ if not _SLUG_RE.fullmatch(expected_id.partition(":")[2]):
145
+ raise ValueError(
146
+ f"code surface path {self.path!r} maps to surface id {expected_id!r}, which "
147
+ "is not a valid 'code:<kebab-slug>' id; use lowercase [a-z0-9] runs separated "
148
+ "by single '/', '.', or '-' characters (for example src/agent-loop.ts)"
149
+ )
150
+ if self.id != expected_id:
151
+ raise ValueError(
152
+ f"surface {self.id!r} does not match its path {self.path!r}: a pathful code "
153
+ f"surface must use id {expected_id!r} (code_surface_id of its path)"
154
+ )
155
+ if self.budget is not None and len(self.content) > self.budget:
156
+ raise ValueError(
157
+ f"surface {self.id!r} content is {len(self.content)} chars, "
158
+ f"over its budget of {self.budget}"
159
+ )
160
+ return self
161
+
162
+ @property
163
+ def slug(self) -> str:
164
+ return self.id.partition(":")[2]
165
+
166
+ @property
167
+ def content_hash(self) -> str:
168
+ return _digest(self.content)
169
+
170
+
171
+ class HarnessDoc(BaseModel):
172
+ """A complete, validated harness: the value the runtime runs and updates are applied to."""
173
+
174
+ name: str
175
+ version: int = Field(default=0, ge=0) # assigned by the store on save; 0 = unsaved
176
+ surfaces: list[Surface]
177
+
178
+ @field_validator("surfaces")
179
+ @classmethod
180
+ def _canonical_order(cls, v: list[Surface]) -> list[Surface]:
181
+ return sorted(v, key=lambda s: s.id)
182
+
183
+ @model_validator(mode="after")
184
+ def _validate_document(self) -> HarnessDoc:
185
+ ids = [s.id for s in self.surfaces]
186
+ duplicates = sorted({i for i in ids if ids.count(i) > 1})
187
+ if duplicates:
188
+ raise ValueError(f"duplicate surface id(s): {duplicates}")
189
+ if not any(s.kind is SurfaceKind.PROMPT for s in self.surfaces):
190
+ raise ValueError("a harness needs at least one prompt surface")
191
+ # These validations construct the derived values; failures surface here, at the boundary.
192
+ self.tools()
193
+ self.max_turns()
194
+ self.max_output_tokens()
195
+ self.temperature()
196
+ for surface in self.surfaces:
197
+ if surface.kind is SurfaceKind.SKILL:
198
+ skill = Skill.from_markdown(surface.content)
199
+ if skill.name != surface.slug:
200
+ raise ValueError(
201
+ f"skill surface {surface.id!r} declares frontmatter name "
202
+ f"{skill.name!r}; the slug and frontmatter name must match"
203
+ )
204
+ elif surface.kind is SurfaceKind.CODE:
205
+ if surface.path is None:
206
+ # The legacy in-process runtime module: a singleton, compile-checked here.
207
+ if surface.id != CODE_RUNTIME_ID:
208
+ raise ValueError(
209
+ f"path-less code surface must be {CODE_RUNTIME_ID!r} "
210
+ f"(got {surface.id!r}); vendored files carry a `path`"
211
+ )
212
+ compile_harness_code(surface.content)
213
+ paths = [s.path for s in self.surfaces if s.path is not None]
214
+ dup_paths = sorted({p for p in paths if paths.count(p) > 1})
215
+ if dup_paths:
216
+ raise ValueError(f"duplicate code surface path(s): {dup_paths}")
217
+ return self
218
+
219
+ # -- surface access ---------------------------------------------------------------------
220
+
221
+ def surface(self, surface_id: str) -> Surface | None:
222
+ for s in self.surfaces:
223
+ if s.id == surface_id:
224
+ return s
225
+ return None
226
+
227
+ def surface_hashes(self) -> dict[str, str]:
228
+ return {s.id: s.content_hash for s in self.surfaces}
229
+
230
+ @property
231
+ def doc_hash(self) -> str:
232
+ """Identity of every surface field that can change materialized execution.
233
+
234
+ Display metadata and validation-only budgets do not affect execution. A code surface's
235
+ destination path does, even when its id and content stay unchanged.
236
+ """
237
+ joined = "\n".join(
238
+ f"{surface.id}\x00{surface.content_hash}"
239
+ + (f"\x00path={surface.path}" if surface.path is not None else "")
240
+ for surface in self.surfaces
241
+ )
242
+ return _digest(joined)
243
+
244
+ @property
245
+ def legacy_doc_hash(self) -> str:
246
+ """The pre-path-inclusion document identity (surface id + content hash only).
247
+
248
+ Exists ONLY so `wmo pull` can integrity-check harness versions the platform recorded
249
+ before `doc_hash` covered materialized paths. Never use it anywhere else: not for new
250
+ records, dedupe, or caching.
251
+ """
252
+ joined = "\n".join(f"{s.id}\x00{s.content_hash}" for s in self.surfaces)
253
+ return _digest(joined)
254
+
255
+ # -- derived, validated views ------------------------------------------------------------
256
+
257
+ def system_prompt(self) -> str:
258
+ """All prompt surfaces joined in id order (a single `prompt:core` is the common case)."""
259
+ parts = [s.content for s in self.surfaces if s.kind is SurfaceKind.PROMPT]
260
+ return "\n\n".join(parts)
261
+
262
+ def tools(self) -> list[str]:
263
+ policy = self.surface(TOOL_POLICY_ID)
264
+ if policy is None:
265
+ return list(DEFAULT_TOOLS)
266
+ names = [line.strip() for line in policy.content.splitlines() if line.strip()]
267
+ resolve_tools(names) # raises on unknown tools / missing submit
268
+ return names
269
+
270
+ def max_turns(self) -> int:
271
+ raw = self.surface(MAX_TURNS_ID)
272
+ if raw is None:
273
+ return DEFAULT_MAX_TURNS
274
+ try:
275
+ value = int(raw.content.strip())
276
+ except ValueError as exc:
277
+ raise ValueError(f"{MAX_TURNS_ID} must be an integer, got {raw.content!r}") from exc
278
+ if value < 1:
279
+ raise ValueError(f"{MAX_TURNS_ID} must be >= 1, got {value}")
280
+ return value
281
+
282
+ def max_output_tokens(self) -> int:
283
+ """Return the per-model-call output cap carried by every pi execution mode."""
284
+ raw = self.surface(MAX_OUTPUT_TOKENS_ID)
285
+ if raw is None:
286
+ return DEFAULT_MAX_OUTPUT_TOKENS
287
+ try:
288
+ value = int(raw.content.strip())
289
+ except ValueError as exc:
290
+ raise ValueError(
291
+ f"{MAX_OUTPUT_TOKENS_ID} must be an integer, got {raw.content!r}"
292
+ ) from exc
293
+ if value < 1:
294
+ raise ValueError(f"{MAX_OUTPUT_TOKENS_ID} must be >= 1, got {value}")
295
+ return value
296
+
297
+ def temperature(self) -> float:
298
+ raw = self.surface(TEMPERATURE_ID)
299
+ if raw is None:
300
+ return DEFAULT_TEMPERATURE
301
+ try:
302
+ value = float(raw.content.strip())
303
+ except ValueError as exc:
304
+ raise ValueError(f"{TEMPERATURE_ID} must be a float, got {raw.content!r}") from exc
305
+ if not 0.0 <= value <= 2.0:
306
+ raise ValueError(f"{TEMPERATURE_ID} must be in [0, 2], got {value}")
307
+ return value
308
+
309
+ def skills(self) -> list[Skill]:
310
+ return [
311
+ Skill.from_markdown(s.content) for s in self.surfaces if s.kind is SurfaceKind.SKILL
312
+ ]
313
+
314
+ def runtime_kind(self) -> str:
315
+ raw = self.surface(RUNTIME_KIND_ID)
316
+ return raw.content.strip() if raw is not None else "kit-python"
317
+
318
+ def code_files(self) -> list[Surface]:
319
+ """The vendored code surfaces (those carrying a file path), in id order."""
320
+ return [s for s in self.surfaces if s.kind is SurfaceKind.CODE and s.path is not None]
321
+
322
+ def runtime(
323
+ self,
324
+ provider: Provider,
325
+ *,
326
+ backend: str = "local",
327
+ e2b_template: str | None = None,
328
+ e2b_pool: E2BSandboxPool | None = None,
329
+ episode_timeout_s: float | None = None,
330
+ context_window: int | None = None,
331
+ transport_retries: int | None = None,
332
+ should_cancel: Callable[[], bool] | None = None,
333
+ ) -> Runtime:
334
+ """The configured agent runtime this document describes.
335
+
336
+ `backend` chooses WHERE the harness process executes; the ENVIRONMENT its tool calls hit
337
+ is whatever `AgentEnvironment` the eval binds (normally the world-model simulation),
338
+ regardless of backend. `local` runs in/from this process. `e2b` runs the harness process
339
+ in E2B sandboxes the runtime owns — only meaningful for `param:runtime-kind` = "pi-node"
340
+ (the vendored pi agent, whose real context management is the point of running it); any
341
+ other kind raises, because its loop already runs in-process and "e2b" would silently mean
342
+ nothing. `e2b_template` names a prebaked sandbox template whose bootstrap (node 22 + pi's
343
+ npm deps) is already done; default is $WMO_E2B_TEMPLATE. Under `local`, "pi-node" uses
344
+ the SSH shim (or the RunnerLink frame transport when PI_TRANSPORT=link); otherwise a
345
+ `code:runtime` surface drives episodes with the harness's own in-process program; with
346
+ neither, the fixed baseline loop runs. `episode_timeout_s` is the pi-node episode wall
347
+ budget, host-enforced on the e2b and link transports and applied as the remote node timeout
348
+ on the SSH transport; omitting it preserves the 300-second default. `context_window` is the
349
+ served context window the runner calibrates pi's context guard to; omitting it asks the
350
+ provider (`ContextWindowProvider`) and otherwise leaves the runner's documented fallback,
351
+ because a wrong window is worse than none. `transport_retries`
352
+ controls whole-episode replay after an E2B transport death; omitting it preserves that
353
+ runtime's one-retry default, and a side-effectful real environment passes 0. Other
354
+ execution modes reject both instead of silently ignoring them. `should_cancel` is
355
+ honored cooperatively by the e2b and link pi-node runtimes; the local SSH pi-node
356
+ runtime has no cancellation hook, so there it is accepted but best-effort only (an
357
+ in-flight episode runs to its own node/SSH bound). All expose the same
358
+ `run(task_id, instruction, environment) -> RunResult` shape closed-loop eval drives.
359
+ """
360
+ if backend not in ("local", "e2b"):
361
+ raise ValueError(f"unknown backend {backend!r}; choose local or e2b")
362
+ if transport_retries is not None and (
363
+ isinstance(transport_retries, bool)
364
+ or not isinstance(transport_retries, int)
365
+ or transport_retries < 0
366
+ ):
367
+ raise ValueError("transport_retries must be a nonnegative integer")
368
+ runtime_kind = self.runtime_kind()
369
+ if episode_timeout_s is not None:
370
+ episode_timeout_s = validate_episode_timeout_s(episode_timeout_s)
371
+ if runtime_kind != "pi-node":
372
+ raise ValueError("episode_timeout_s applies only to pi-node execution")
373
+ if context_window is not None and runtime_kind != "pi-node":
374
+ raise ValueError("context_window applies only to pi-node execution")
375
+ if transport_retries is not None and (backend != "e2b" or runtime_kind != "pi-node"):
376
+ raise ValueError("transport_retries applies only to e2b pi-node execution")
377
+ if runtime_kind == "pi-node":
378
+ skills = SkillLibrary(self.skills())
379
+ code_files = {s.path: s.content for s in self.code_files() if s.path is not None}
380
+ tool_names = self.tools()
381
+ # Progressive disclosure is runtime plumbing, not a burden on every persisted tool
382
+ # policy. Keep pi-node behavior aligned with AgentRuntime: a skill-bearing document
383
+ # always exposes read_skill, even when the authored policy lists only env tools.
384
+ if len(skills) and READ_SKILL.name not in tool_names:
385
+ tool_names.append(READ_SKILL.name)
386
+ tools = resolve_tools(tool_names)
387
+ structured_provider = provider if isinstance(provider, ToolCallingProvider) else None
388
+ if (
389
+ backend == "e2b" or os.environ.get("PI_TRANSPORT") == "link"
390
+ ) and structured_provider is None:
391
+ raise TypeError(
392
+ "pi-node link/e2b execution needs a ToolCallingProvider; "
393
+ "use a structured provider or WaterfallProvider"
394
+ )
395
+ if backend == "e2b":
396
+ # Lazy: the e2b backend is an optional extra; `local` must import none of it.
397
+ from wmo.harness.pi_e2b import E2BPiRuntime
398
+
399
+ assert structured_provider is not None
400
+ return E2BPiRuntime(
401
+ provider=structured_provider,
402
+ files=code_files,
403
+ tools=tools,
404
+ system_prompt=self.assembled_prompt(skills, structured_tools=True),
405
+ temperature=self.temperature(),
406
+ skills=skills,
407
+ template=e2b_template,
408
+ pool=e2b_pool,
409
+ max_turns=self.max_turns(),
410
+ max_output_tokens=self.max_output_tokens(),
411
+ episode_timeout_s=(
412
+ DEFAULT_EVAL_EPISODE_TIMEOUT_S
413
+ if episode_timeout_s is None
414
+ else episode_timeout_s
415
+ ),
416
+ context_window=context_window,
417
+ transport_retries=(1 if transport_retries is None else transport_retries),
418
+ should_cancel=should_cancel,
419
+ )
420
+ # PI_TRANSPORT=link routes pi to the RunnerLink frame transport (a persistent runner the
421
+ # host set via runner_link.set_active_channel) instead of the per-episode SSH shim; the
422
+ # default (unset / "ssh") keeps PiRuntime. The worker LLM reads the same PI_AGENT_* env.
423
+ if os.environ.get("PI_TRANSPORT") == "link":
424
+ from wmo.harness.runner_link import (
425
+ RunnerLink,
426
+ active_channel,
427
+ )
428
+
429
+ channel = active_channel()
430
+ if channel is None:
431
+ raise RuntimeError(
432
+ "PI_TRANSPORT=link but no active runner channel; call "
433
+ "runner_link.set_active_channel(channel) before running episodes"
434
+ )
435
+ assert structured_provider is not None
436
+ return RunnerLink(
437
+ channel,
438
+ tools=tools,
439
+ provider=structured_provider,
440
+ system_prompt=self.assembled_prompt(skills, structured_tools=True),
441
+ files=code_files,
442
+ temperature=self.temperature(),
443
+ skills=skills,
444
+ max_turns=self.max_turns(),
445
+ max_output_tokens=self.max_output_tokens(),
446
+ episode_timeout_s=episode_timeout_s,
447
+ context_window=context_window,
448
+ should_cancel=should_cancel,
449
+ )
450
+ from wmo.harness.pi_runtime import PiRuntime # circular: pi_runtime imports doc
451
+
452
+ return PiRuntime(
453
+ provider,
454
+ files=code_files,
455
+ tools=tools,
456
+ temperature=self.temperature(),
457
+ skills=skills,
458
+ system_prompt=self.assembled_prompt(skills, structured_tools=True),
459
+ max_turns=self.max_turns(),
460
+ max_output_tokens=self.max_output_tokens(),
461
+ episode_timeout_s=(
462
+ DEFAULT_EVAL_EPISODE_TIMEOUT_S
463
+ if episode_timeout_s is None
464
+ else episode_timeout_s
465
+ ),
466
+ context_window=context_window,
467
+ )
468
+ if backend == "e2b":
469
+ raise ValueError(
470
+ f"backend='e2b' runs the pi-node harness process in a sandbox; this harness's "
471
+ f"runtime kind is {self.runtime_kind()!r}, which already runs in-process — "
472
+ "use backend='local'"
473
+ )
474
+ code = self.surface(CODE_RUNTIME_ID)
475
+ skills = SkillLibrary(self.skills())
476
+ if code is not None:
477
+ return CodeRuntime(
478
+ provider,
479
+ code=code.content,
480
+ tools=resolve_tools(self.tools()),
481
+ temperature=self.temperature(),
482
+ skills=skills,
483
+ system_prompt=self.assembled_prompt(skills),
484
+ )
485
+ return AgentRuntime(
486
+ provider,
487
+ system_prompt=self.system_prompt(),
488
+ tools=self.tools(),
489
+ max_turns=self.max_turns(),
490
+ temperature=self.temperature(),
491
+ skills=skills,
492
+ )
493
+
494
+ def assembled_prompt(
495
+ self, skills: SkillLibrary | None = None, *, structured_tools: bool = False
496
+ ) -> str:
497
+ """Return the system prompt shared by episode and project session runtimes.
498
+
499
+ Args:
500
+ skills: The resolved skill library; defaults to this document's skills.
501
+ structured_tools: True when the runtime passes STRUCTURED tool schemas to the model and
502
+ its own renderer defines the calling convention (every pi runtime). The
503
+ JSON-action clause is then dropped: carrying it declares a protocol that is not in
504
+ use, and in the Nemotron-3 runs 14.7% of trials spent reasoning on a JSON envelope
505
+ the renderer never parses. Only the runtimes that really call `parse_tool_call`
506
+ (`AgentRuntime`, `CodeRuntime`) leave it in.
507
+
508
+ Returns:
509
+ The assembled system prompt.
510
+ """
511
+ resolved_skills = skills if skills is not None else SkillLibrary(self.skills())
512
+ tool_names = self.tools()
513
+ if len(resolved_skills) and READ_SKILL.name not in tool_names:
514
+ tool_names.append(READ_SKILL.name)
515
+ core = self.system_prompt()
516
+ if structured_tools:
517
+ core = strip_json_protocol_clause(core)
518
+ prompt = f"{core}\n\n## Tools\n{render_tools(resolve_tools(tool_names))}"
519
+ index = resolved_skills.render_index()
520
+ if index:
521
+ prompt += f"\n\n## Your skills (read a body with read_skill)\n{index}"
522
+ return prompt
523
+
524
+ @classmethod
525
+ def baseline(cls, name: str = "baseline") -> HarnessDoc:
526
+ """The default harness: one core prompt, the default tools, default loop params."""
527
+ return cls(
528
+ name=name,
529
+ surfaces=[
530
+ Surface(id="prompt:core", kind=SurfaceKind.PROMPT, content=DEFAULT_SYSTEM_PROMPT),
531
+ Surface(
532
+ id=TOOL_POLICY_ID,
533
+ kind=SurfaceKind.TOOL_POLICY,
534
+ content="\n".join(DEFAULT_TOOLS),
535
+ ),
536
+ Surface(id=MAX_TURNS_ID, kind=SurfaceKind.PARAM, content=str(DEFAULT_MAX_TURNS)),
537
+ Surface(
538
+ id=TEMPERATURE_ID, kind=SurfaceKind.PARAM, content=str(DEFAULT_TEMPERATURE)
539
+ ),
540
+ ],
541
+ )
542
+
543
+
544
+ def code_baseline(name: str = "baseline") -> HarnessDoc:
545
+ """The baseline harness with its loop as an editable `code:runtime` surface.
546
+
547
+ Behaviorally equivalent to `HarnessDoc.baseline()` — same prompt, tools, and one-call-per-turn
548
+ loop, but the loop is data, so `wmo optimize` can propose structural changes to it.
549
+ """
550
+ base = HarnessDoc.baseline(name)
551
+ code = Surface(id=CODE_RUNTIME_ID, kind=SurfaceKind.CODE, content=DEFAULT_RUNTIME_CODE)
552
+ return HarnessDoc(name=name, surfaces=[*base.surfaces, code])
553
+
554
+
555
+ def _digest(text: str) -> str:
556
+ return hashlib.blake2b(text.encode("utf-8"), digest_size=16).hexdigest()