world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,495 @@
1
+ """`PiRuntime`: run the vendored pi agent (a real multi-file TypeScript harness) as an episode.
2
+
3
+ The harness under search is the pi agent's own source: each file is a `code:` surface carrying a
4
+ `path`. To run one task the runtime materializes those files into a checkout on a runner box,
5
+ starts a local shim, and drives pi headless through it (`wmo/harness/pi_entry/entry.ts`):
6
+
7
+ - pi's LLM calls hit the shim's OpenAI-compatible `/v1/chat/completions`; the shim validates the
8
+ structured request and delegates it to the caller's tool-calling provider. Provider-owned auth,
9
+ routing, translation, retries, and waterfall failover stay on the control host.
10
+ - pi's task tools POST `/tool`, which the runtime answers from the `AgentEnvironment` (the world
11
+ model in simulation, the real backend in the transfer check). These calls are the recorded
12
+ transcript the judge grades.
13
+ - `submit` POSTs `/done`; the runtime returns a `RunResult` shaped exactly like the other runtimes.
14
+
15
+ The runner is remote (node lives on a separate box, never the control host), reached over SSH with
16
+ a reverse tunnel so the runner's node process can call back to the shim. The environment budget is
17
+ enforced kit-style: past the cap, `/tool` returns an error observation and the episode ends.
18
+
19
+ Concurrency note: episodes are serialized on one runner directory + port. Parallel rollouts must
20
+ pass distinct `port`/`workdir` (a per-episode caller responsibility); the default is a single
21
+ sequential lane, which is what the current search driver uses.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ import json
27
+ import os
28
+ import re
29
+ import subprocess
30
+ import threading
31
+ from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
32
+
33
+ from llm_waterfall import ChatRequest
34
+ from pydantic import JsonValue
35
+
36
+ from wmo.core.types import Action, ActionKind, EnvState, JsonObject, Observation, Step
37
+ from wmo.harness.environment import AgentEnvironment, is_env_action
38
+ from wmo.harness.runner_link import params_schema, provider_context_window, stop_reason_for_done
39
+ from wmo.harness.runtime import (
40
+ DEFAULT_EVAL_EPISODE_TIMEOUT_S,
41
+ DEFAULT_MAX_OUTPUT_TOKENS,
42
+ DEFAULT_MAX_TURNS,
43
+ RunResult,
44
+ StopReason,
45
+ validate_episode_timeout_s,
46
+ )
47
+ from wmo.harness.skills import SkillLibrary
48
+ from wmo.harness.tools import READ_SKILL, ToolSpec
49
+ from wmo.providers.base import UNPARSED_TOOL_CALLS_KEY, Provider, ToolCallingProvider
50
+
51
+ # The runner: node runs here, reached over SSH. The checkout keeps pi's node_modules; per-episode
52
+ # source is overwritten from the harness surfaces.
53
+ PI_RUNNER_HOST = os.environ.get("PI_RUNNER_HOST", "kion@nucbox.local")
54
+ PI_RUNNER_DIR = os.environ.get("PI_RUNNER_DIR", "~/pi-run")
55
+ DEFAULT_MAX_ENV_ACTIONS = 40
56
+ _PI_ENTRY_DIR = os.path.join(os.path.dirname(__file__), "pi_entry")
57
+ _ENTRY_TS = os.path.join(_PI_ENTRY_DIR, "entry.ts")
58
+ # entry.ts imports the shared classify + nudge policy, so it must be materialized alongside it.
59
+ _TERMINATION_TS = os.path.join(_PI_ENTRY_DIR, "runner_termination.ts")
60
+ # Cleanup headroom past the node wall budget: one SSH round trip plus process teardown.
61
+ _NODE_TEARDOWN_GRACE_S = 60.0
62
+ # Runner paths are interpolated into remote shell commands, so restrict them to characters that
63
+ # cannot break out of the command (allows `~` expansion; rejects spaces, quotes, `;`, `$`, etc.).
64
+ _SAFE_REMOTE_PATH = re.compile(r"^[A-Za-z0-9_./~-]+$")
65
+
66
+
67
+ class _MaterializeError(RuntimeError):
68
+ """Remote source materialization failed; the episode must not run stale files."""
69
+
70
+
71
+ class _Episode:
72
+ """Mutable per-run state the shim handlers share."""
73
+
74
+ def __init__(
75
+ self,
76
+ *,
77
+ instruction: str,
78
+ system_prompt: str,
79
+ tools: list[ToolSpec],
80
+ provider: ToolCallingProvider,
81
+ environment: AgentEnvironment,
82
+ temperature: float,
83
+ skills: SkillLibrary,
84
+ max_env_actions: int,
85
+ max_turns: int,
86
+ max_output_tokens: int,
87
+ context_window: int | None = None,
88
+ ) -> None:
89
+ self.instruction = instruction
90
+ self.system_prompt = system_prompt
91
+ self.tools = tools
92
+ self.provider = provider
93
+ self.environment = environment
94
+ self.temperature = temperature
95
+ self.skills = skills
96
+ self.max_env_actions = max_env_actions
97
+ self.max_turns = max_turns
98
+ self.max_output_tokens = max_output_tokens
99
+ self.context_window = context_window
100
+ self.steps: list[Step] = []
101
+ self.answer: str = ""
102
+ self.proxy_error: str = ""
103
+ self.done_reason: str = ""
104
+ # The host's view of the most recent worker completion, which entry.ts reads back through
105
+ # GET /signal so its termination classifier sees the same evidence the frame runners get.
106
+ self.finish_reason: str = ""
107
+ self.unparsed_tool_calls: list[str] = []
108
+ self.tool_call_turns: int = 0
109
+ self.done = threading.Event()
110
+ self._env_calls = 0
111
+
112
+ def task_json(self) -> JsonObject:
113
+ return {
114
+ "instruction": self.instruction,
115
+ "system": self.system_prompt,
116
+ "max_turns": self.max_turns,
117
+ "max_output_tokens": self.max_output_tokens,
118
+ "context_window": self.context_window,
119
+ "tools": [
120
+ {"name": t.name, "description": t.description, "parameters": _params_schema(t)}
121
+ for t in self.tools
122
+ ],
123
+ }
124
+
125
+ def signal_json(self) -> JsonObject:
126
+ """The host's view of the last worker completion, for entry.ts's classifier."""
127
+ return {
128
+ "finish_reason": self.finish_reason,
129
+ "unparsed_tool_calls": list(self.unparsed_tool_calls),
130
+ "provider_error": self.proxy_error,
131
+ "tool_call_turns": self.tool_call_turns,
132
+ }
133
+
134
+ def run_tool(self, name: str, arguments: JsonObject) -> JsonObject:
135
+ action = Action(kind=ActionKind.TOOL_CALL, name=name, arguments=arguments)
136
+ if name not in {t.name for t in self.tools}:
137
+ obs = Observation(content=f"tool {name!r} not available", is_error=True)
138
+ elif name == READ_SKILL.name:
139
+ raw_name = arguments.get("name")
140
+ skill_name = raw_name if isinstance(raw_name, str) else ""
141
+ skill = self.skills.get(skill_name)
142
+ if skill is None:
143
+ obs = Observation(content=f"no skill named {skill_name!r}", is_error=True)
144
+ else:
145
+ obs = Observation(content=skill.body)
146
+ elif self._env_calls >= self.max_env_actions:
147
+ obs = Observation(content="environment action budget exhausted", is_error=True)
148
+ elif not is_env_action(action):
149
+ obs = Observation(content=f"tool {name!r} not available", is_error=True)
150
+ else:
151
+ self._env_calls += 1
152
+ obs = self.environment.execute(action)
153
+ self.steps.append(
154
+ Step(action=action, observation=obs, state_before=EnvState(), task=self.instruction)
155
+ )
156
+ return {"content": obs.content, "is_error": obs.is_error}
157
+
158
+ def worker_request(self, body: JsonObject) -> ChatRequest:
159
+ """Apply the document sampling policy to one runner-authored structured request."""
160
+ request_body = dict(body)
161
+ request_body["temperature"] = self.temperature
162
+ return ChatRequest.model_validate(request_body)
163
+
164
+
165
+ # The tool `parameters` schema builder lives in runner_link (shared with the frame transport);
166
+ # re-exported here under its old private name so existing callers and tests keep working.
167
+ _params_schema = params_schema
168
+
169
+
170
+ class _ShimServer(ThreadingHTTPServer):
171
+ """A threading HTTP server that carries the current episode for its handlers."""
172
+
173
+ # Environment calls mutate evaluator-owned state, so server_close() must join every active
174
+ # handler: no environment write may land after the episode is declared finished (against a
175
+ # real execution environment a late write would mutate state the evaluator is already
176
+ # verifying; with the world model it was merely cosmetic). The join is bounded by whatever
177
+ # the slowest handler is blocked on, worst case a completion handler waiting out the provider
178
+ # SDK's timeout and retries (minutes during an outage), not just a tool command's budget. A
179
+ # slow close is the accepted price of a trustworthy verdict.
180
+ daemon_threads = False
181
+ episode: _Episode
182
+
183
+
184
+ class _ShimHandler(BaseHTTPRequestHandler):
185
+ # HTTP/1.1 so the OpenAI SDK's keep-alive works; the SSE handler forces a fresh socket per
186
+ # turn (see _serve_completion) to avoid mis-framing the pipelined next request.
187
+ protocol_version = "HTTP/1.1"
188
+
189
+ def log_message(self, format: str, *args: object) -> None: # noqa: A002 - base API name
190
+ return # silence per-request stderr spam
191
+
192
+ @property
193
+ def _ep(self) -> _Episode:
194
+ assert isinstance(self.server, _ShimServer)
195
+ return self.server.episode
196
+
197
+ def _read_body(self) -> JsonObject:
198
+ length = int(self.headers.get("Content-Length", 0))
199
+ raw = self.rfile.read(length) if length else b"{}"
200
+ return json.loads(raw or b"{}")
201
+
202
+ def _send_json(self, obj: JsonObject, status: int = 200) -> None:
203
+ body = json.dumps(obj).encode("utf-8")
204
+ self.send_response(status)
205
+ self.send_header("Content-Type", "application/json")
206
+ self.send_header("Content-Length", str(len(body)))
207
+ self.end_headers()
208
+ self.wfile.write(body)
209
+
210
+ def do_GET(self) -> None: # noqa: N802 - BaseHTTPRequestHandler API
211
+ path = self.path.rstrip("/")
212
+ if path == "/task":
213
+ self._send_json(self._ep.task_json())
214
+ elif path == "/signal":
215
+ self._send_json(self._ep.signal_json())
216
+ else:
217
+ self._send_json({"error": "not found"}, status=404)
218
+
219
+ def do_POST(self) -> None: # noqa: N802 - BaseHTTPRequestHandler API
220
+ path = self.path.rstrip("/")
221
+ if path == "/v1/chat/completions":
222
+ self._serve_completion(self._read_body())
223
+ elif path == "/tool":
224
+ body = self._read_body()
225
+ name = body.get("name")
226
+ args = body.get("arguments")
227
+ self._send_json(
228
+ self._ep.run_tool(
229
+ name if isinstance(name, str) else "",
230
+ args if isinstance(args, dict) else {},
231
+ )
232
+ )
233
+ elif path == "/done":
234
+ body = self._read_body()
235
+ answer = body.get("answer")
236
+ reason = body.get("reason")
237
+ self._ep.answer = answer if isinstance(answer, str) else ""
238
+ self._ep.done_reason = reason if isinstance(reason, str) else ""
239
+ self._send_json({})
240
+ self._ep.done.set()
241
+ else:
242
+ self._send_json({"error": "not found"}, status=404)
243
+
244
+ def _serve_completion(self, body: JsonObject) -> None:
245
+ """Delegate pi's structured request to the provider and synthesize OpenAI SSE."""
246
+ self.send_response(200)
247
+ self.send_header("Content-Type", "text/event-stream")
248
+ self.send_header("Connection", "close")
249
+ self.end_headers()
250
+ self.close_connection = True
251
+ try:
252
+ completion = self._ep.provider.complete_chat(self._ep.worker_request(body))
253
+ choice = completion.choices[0]
254
+ message = choice.message
255
+ # Record the host's view of this turn BEFORE streaming it: entry.ts reads it back to
256
+ # classify why the episode ended (truncated at the cap vs unparsed vs prose only).
257
+ self._ep.proxy_error = ""
258
+ self._ep.finish_reason = choice.finish_reason or "stop"
259
+ unparsed = (choice.model_extra or {}).get(UNPARSED_TOOL_CALLS_KEY)
260
+ self._ep.unparsed_tool_calls = (
261
+ [str(item) for item in unparsed] if isinstance(unparsed, list) else []
262
+ )
263
+ if message.tool_calls:
264
+ self._ep.tool_call_turns += 1
265
+ content = message.content if isinstance(message.content, str) else ""
266
+ delta: dict[str, JsonValue] = {"role": "assistant", "content": content}
267
+ if message.tool_calls:
268
+ delta["tool_calls"] = [
269
+ {
270
+ "index": i,
271
+ **tool_call.model_dump(mode="json"),
272
+ }
273
+ for i, tool_call in enumerate(message.tool_calls)
274
+ ]
275
+ first = {"choices": [{"index": 0, "delta": delta, "finish_reason": None}]}
276
+ last = {
277
+ "choices": [
278
+ {"index": 0, "delta": {}, "finish_reason": choice.finish_reason or "stop"}
279
+ ]
280
+ }
281
+ self.wfile.write(f"data: {json.dumps(first)}\n\n".encode())
282
+ self.wfile.write(f"data: {json.dumps(last)}\n\n".encode())
283
+ self.wfile.write(b"data: [DONE]\n\n")
284
+ except Exception as exc: # noqa: BLE001 - never crash the shim
285
+ self._ep.proxy_error = str(exc)
286
+ err = json.dumps({"error": {"message": f"agent provider failed: {exc}"}})
287
+ self.wfile.write(f"data: {err}\n\ndata: [DONE]\n\n".encode())
288
+
289
+
290
+ class PiRuntime:
291
+ """Runs one episode of the vendored pi harness against an `AgentEnvironment`."""
292
+
293
+ def __init__(
294
+ self,
295
+ provider: Provider,
296
+ *,
297
+ files: dict[str, str],
298
+ tools: list[ToolSpec],
299
+ temperature: float = 0.7,
300
+ skills: SkillLibrary | None = None,
301
+ system_prompt: str = "",
302
+ port: int = 8891,
303
+ workdir: str | None = None,
304
+ max_env_actions: int = DEFAULT_MAX_ENV_ACTIONS,
305
+ max_turns: int = DEFAULT_MAX_TURNS,
306
+ max_output_tokens: int = DEFAULT_MAX_OUTPUT_TOKENS,
307
+ episode_timeout_s: float = DEFAULT_EVAL_EPISODE_TIMEOUT_S,
308
+ context_window: int | None = None,
309
+ ) -> None:
310
+ if not isinstance(provider, ToolCallingProvider):
311
+ raise TypeError("PiRuntime needs a ToolCallingProvider")
312
+ self._provider = provider
313
+ self._files = files
314
+ self._skills = skills if skills is not None else SkillLibrary()
315
+ self._tools = list(tools)
316
+ if len(self._skills) and READ_SKILL.name not in {tool.name for tool in self._tools}:
317
+ self._tools.append(READ_SKILL)
318
+ if not 0.0 <= temperature <= 2.0:
319
+ raise ValueError("temperature must be in [0, 2]")
320
+ self._temperature = temperature
321
+ self._system_prompt = system_prompt
322
+ self._port = port
323
+ self._workdir = workdir or f"{PI_RUNNER_DIR}/ep-{port}"
324
+ self._max_env_actions = max_env_actions
325
+ if max_turns < 1:
326
+ raise ValueError("max_turns must be >= 1")
327
+ if max_output_tokens < 1:
328
+ raise ValueError("max_output_tokens must be >= 1")
329
+ self._max_turns = max_turns
330
+ self._max_output_tokens = max_output_tokens
331
+ # The SSH path used to hardcode `timeout 300 node`, so every configured wall budget was
332
+ # silently 300s and 30% of long TerminalBench-2 trials died on it.
333
+ self._episode_timeout_s = validate_episode_timeout_s(episode_timeout_s)
334
+ self._context_window = (
335
+ context_window if context_window is not None else provider_context_window(provider)
336
+ )
337
+ for label, path in (("PI_RUNNER_DIR", PI_RUNNER_DIR), ("workdir", self._workdir)):
338
+ if not _SAFE_REMOTE_PATH.match(path):
339
+ raise ValueError(
340
+ f"unsafe remote {label} {path!r}: only [A-Za-z0-9_./~-] allowed "
341
+ "(it is interpolated into a remote shell command)"
342
+ )
343
+
344
+ def run(self, task_id: str, instruction: str, environment: AgentEnvironment) -> RunResult:
345
+ episode = _Episode(
346
+ instruction=instruction,
347
+ system_prompt=self._system_prompt,
348
+ tools=self._tools,
349
+ provider=self._provider,
350
+ environment=environment,
351
+ temperature=self._temperature,
352
+ skills=self._skills,
353
+ max_env_actions=self._max_env_actions,
354
+ max_turns=self._max_turns,
355
+ max_output_tokens=self._max_output_tokens,
356
+ context_window=self._context_window,
357
+ )
358
+ server = _ShimServer(("127.0.0.1", self._port), _ShimHandler)
359
+ server.episode = episode
360
+ thread = threading.Thread(target=server.serve_forever, daemon=True)
361
+ thread.start()
362
+ try:
363
+ try:
364
+ self._materialize()
365
+ except _MaterializeError as exc:
366
+ # Remote write failed; do not run node against stale files from a prior episode.
367
+ return self._error_result(task_id, episode, instruction, str(exc), StopReason.ERROR)
368
+ code, note = self._run_node()
369
+ finally:
370
+ server.shutdown()
371
+ server.server_close()
372
+ if not episode.done.is_set():
373
+ stop = StopReason.ERROR if code != 0 else StopReason.MAX_TURNS
374
+ return self._error_result(
375
+ task_id, episode, instruction, note or "episode ended without submit", stop
376
+ )
377
+ if episode.proxy_error:
378
+ # The worker LLM proxy failed (auth/outage/HTTP error); entry.ts still POSTs /done, but
379
+ # this is infrastructure failure, not an agent submission, so never count it as
380
+ # SUBMITTED.
381
+ return self._error_result(
382
+ task_id,
383
+ episode,
384
+ instruction,
385
+ f"worker LLM proxy error: {episode.proxy_error}",
386
+ StopReason.PROVIDER_ERROR,
387
+ )
388
+ # entry.ts reports WHY it finished; only an explicit submit is a completion.
389
+ stop_reason = stop_reason_for_done(episode.done_reason)
390
+ return RunResult(
391
+ task_id=task_id,
392
+ steps=episode.steps,
393
+ stop_reason=stop_reason,
394
+ answer=episode.answer,
395
+ turns=len(episode.steps),
396
+ )
397
+
398
+ @staticmethod
399
+ def _error_result(
400
+ task_id: str, episode: _Episode, instruction: str, note: str, stop: StopReason
401
+ ) -> RunResult:
402
+ episode.steps.append(
403
+ Step(
404
+ action=Action(kind=ActionKind.MESSAGE, content="(pi runtime)"),
405
+ observation=Observation(content=note, is_error=True),
406
+ state_before=EnvState(),
407
+ task=instruction,
408
+ )
409
+ )
410
+ return RunResult(
411
+ task_id=task_id,
412
+ steps=episode.steps,
413
+ stop_reason=stop,
414
+ answer="",
415
+ turns=len(episode.steps),
416
+ )
417
+
418
+ def _materialize(self) -> None:
419
+ """Write the harness's code surfaces + entry.ts into the runner checkout via SSH.
420
+
421
+ The files stream as one JSON blob into a python materializer on the runner (one SSH round
422
+ trip, no per-file scp), with node_modules symlinked from the persistent checkout.
423
+ """
424
+ blob = json.dumps(
425
+ {
426
+ "entry.ts": _read(_ENTRY_TS),
427
+ "runner_termination.ts": _read(_TERMINATION_TS),
428
+ **self._files,
429
+ }
430
+ )
431
+ writer = (
432
+ "import json,sys,os\n"
433
+ "d=json.load(sys.stdin)\n"
434
+ "for p,c in d.items():\n"
435
+ " os.makedirs(os.path.dirname(p) or '.',exist_ok=True)\n"
436
+ " open(p,'w').write(c)\n"
437
+ )
438
+ remote = (
439
+ f"mkdir -p {self._workdir}"
440
+ f" && ln -sfn {PI_RUNNER_DIR}/node_modules {self._workdir}/node_modules"
441
+ f" && cd {self._workdir} && python3 -c {_shq(writer)}"
442
+ )
443
+ result = _ssh(remote, input_bytes=blob.encode("utf-8"))
444
+ if result.returncode != 0:
445
+ detail = (result.stderr or b"").decode("utf-8", "replace").strip()[-300:]
446
+ raise _MaterializeError(f"remote materialize failed (rc={result.returncode}): {detail}")
447
+
448
+ def _run_node(self) -> tuple[int, str]:
449
+ """Run entry.ts on the runner with a reverse tunnel back to the local shim.
450
+
451
+ The node wall budget is the configured episode timeout, not a fixed 300s: TerminalBench-2
452
+ tasks compile toolchains and boot VMs, and the old constant killed 30% of them mid-turn.
453
+ """
454
+ url = f"http://127.0.0.1:{self._port}"
455
+ node_timeout_s = int(self._episode_timeout_s)
456
+ remote_cmd = (
457
+ f"cd {self._workdir} && PI_SHIM_URL={url} "
458
+ f"timeout {node_timeout_s} node --experimental-strip-types entry.ts"
459
+ )
460
+ proc = subprocess.run(
461
+ [
462
+ "ssh",
463
+ "-o",
464
+ "ConnectTimeout=10",
465
+ "-o",
466
+ "BatchMode=yes",
467
+ "-R",
468
+ f"{self._port}:127.0.0.1:{self._port}",
469
+ PI_RUNNER_HOST,
470
+ remote_cmd,
471
+ ],
472
+ capture_output=True,
473
+ text=True,
474
+ timeout=self._episode_timeout_s + _NODE_TEARDOWN_GRACE_S,
475
+ )
476
+ return proc.returncode, (proc.stderr or "").strip()[-500:]
477
+
478
+
479
+ def _ssh(remote_cmd: str, input_bytes: bytes | None = None) -> subprocess.CompletedProcess[bytes]:
480
+ return subprocess.run(
481
+ ["ssh", "-o", "ConnectTimeout=10", "-o", "BatchMode=yes", PI_RUNNER_HOST, remote_cmd],
482
+ input=input_bytes,
483
+ capture_output=True,
484
+ timeout=120,
485
+ )
486
+
487
+
488
+ def _read(path: str) -> str:
489
+ with open(path, encoding="utf-8") as fh:
490
+ return fh.read()
491
+
492
+
493
+ def _shq(text: str) -> str:
494
+ """Single-quote a string for a remote shell (the python -c body)."""
495
+ return "'" + text.replace("'", "'\\''") + "'"
@@ -0,0 +1,65 @@
1
+ """The vendored pi agent, and the seam that turns it into a searchable harness.
2
+
3
+ `wmo/harness/vendor/pi-agent/` is a byte-exact copy of `packages/agent` from
4
+ earendil-works/pi at v0.80.3 (commit a23abe4a695df8b69b613f73e9fdda2a8af894d4). The pin, the
5
+ license attribution, and the integrity ledger live beside it: `vendor/pi-agent/VENDOR.md`,
6
+ `vendor/pi-agent/LICENSE`, and `vendor/manifest.sha256` (regenerate/verify with
7
+ `wmo/harness/vendor/vendor_pi.sh`).
8
+
9
+ This module is the ONLY place wmo reads that tree, and it reads it straight from disk:
10
+ `pi_agent_code_surfaces()` loads pi's own TypeScript source into `code:` surfaces so the
11
+ meta-agent searches over the real agent's source, and `PiRuntime` materializes those surfaces to
12
+ run pi headless. Nothing here fetches pi over the network or from a scratch checkout — the
13
+ committed vendored copy is the sole source of truth. The whole 56-file package is vendored on disk
14
+ (byte-checked against upstream); the 25 runnable `src/**/*.ts` files become the searchable
15
+ surfaces (fixtures, docs, and the package's own vitest specs are vendored but not surfaced).
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ from pathlib import Path
21
+
22
+ # code_surface_id lives with the Surface grammar in doc.py; imported (and re-exported) here for
23
+ # the existing pi-vendor call sites.
24
+ from wmo.harness.doc import Surface, SurfaceKind, code_surface_id
25
+
26
+ # The committed, byte-exact vendored copy (see VENDOR.md for the upstream pin).
27
+ PI_AGENT_ROOT = Path(__file__).parent / "vendor" / "pi-agent"
28
+ # pi's runnable TypeScript source — the harness the meta-agent searches over.
29
+ _SOURCE_GLOB = "src/**/*.ts"
30
+
31
+
32
+ def pi_agent_source_paths() -> list[Path]:
33
+ """Every runnable pi source file under the vendored tree, sorted, excluding vitest specs."""
34
+ return sorted(
35
+ p
36
+ for p in PI_AGENT_ROOT.glob(_SOURCE_GLOB)
37
+ if p.is_file() and not p.name.endswith(".test.ts")
38
+ )
39
+
40
+
41
+ def pi_agent_code_surfaces() -> list[Surface]:
42
+ """pi's vendored source as `code:` surfaces, each carrying its path under the package root.
43
+
44
+ Read straight from `PI_AGENT_ROOT` (the committed copy) — never from a network fetch or a
45
+ scratch checkout. Raises if the vendored tree is missing so a broken vendoring fails loudly
46
+ instead of silently running an empty harness.
47
+ """
48
+ paths = pi_agent_source_paths()
49
+ if not paths:
50
+ raise FileNotFoundError(
51
+ f"no pi source under {PI_AGENT_ROOT}; is the vendored copy present? "
52
+ "regenerate with wmo/harness/vendor/vendor_pi.sh"
53
+ )
54
+ surfaces: list[Surface] = []
55
+ for p in paths:
56
+ rel = p.relative_to(PI_AGENT_ROOT).as_posix()
57
+ surfaces.append(
58
+ Surface(
59
+ id=code_surface_id(rel),
60
+ kind=SurfaceKind.CODE,
61
+ path=rel,
62
+ content=p.read_text(encoding="utf-8"),
63
+ )
64
+ )
65
+ return surfaces