world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,142 @@
1
+ /**
2
+ * Why a pi episode finished, and what to do about it before giving up.
3
+ *
4
+ * pi's `agent.prompt()` resolves as soon as an assistant turn carries no tool calls. That single
5
+ * event covers four completely different things: the model called `submit`, it emitted prose, its
6
+ * tool call was cut off at the output-token cap, or the renderer could not parse the tool call it
7
+ * did emit. The episode runners used to report all four as one `done` frame, which the host mapped
8
+ * to `submitted`, so every one of them scored as a clean completion with reward 0.
9
+ *
10
+ * This module is the shared classifier + nudge policy the three episode runners
11
+ * (`runner_stdio.ts`, `runner_service.ts`, `entry.ts`) drive:
12
+ *
13
+ * - `TurnSignal` is updated by each runner's LLM bridge from the host's completion frame, so the
14
+ * runner can see the finish_reason, whether tool calls came back, and any tool-call parse
15
+ * errors the host's renderer reported (`wmo_unparsed_tool_calls`).
16
+ * - `classifyEnd` turns that plus the runner's own flags into a `DoneReason`.
17
+ * - `nudgeFor` is the observation fed back to the model instead of ending the episode, modeled
18
+ * on the reference terminus-2 agent's behavior (report the parser's complaint, tell the model
19
+ * to re-issue in smaller chunks, and ask it to either act or submit).
20
+ * - `shouldNudge` bounds that to MAX_NONACTION_TURNS consecutive non-action turns.
21
+ *
22
+ * `DoneReason` values are the wire vocabulary `wmo/harness/runner_link.py` maps onto distinct
23
+ * `StopReason`s, and MAX_NONACTION_TURNS mirrors `wmo.harness.runtime.MAX_NONACTION_TURNS`.
24
+ */
25
+
26
+ /** The `done` frame's `reason`: exactly why this episode stopped. */
27
+ export type DoneReason =
28
+ | "submit"
29
+ | "no_tool_call"
30
+ | "output_truncated"
31
+ | "unparsed_tool_call"
32
+ | "provider_error"
33
+ | "max_turns";
34
+
35
+ /** Consecutive non-action turns a runner nudges through before reporting `done`. */
36
+ export const MAX_NONACTION_TURNS = 3;
37
+
38
+ /** What the last host completion frame carried, as the bridge observed it. */
39
+ export interface TurnSignal {
40
+ /** finish_reason of the most recent completion ("length" means the output cap was hit). */
41
+ finishReason: string;
42
+ /** Tool-call parse errors the host's renderer reported for the most recent completion. */
43
+ unparsedToolCallErrors: string[];
44
+ /** The most recent host-side worker error (context overflow, outage), or "". */
45
+ providerError: string;
46
+ /** Cumulative count of completions that carried at least one tool call. */
47
+ toolCallTurns: number;
48
+ }
49
+
50
+ /** A fresh signal, before any completion has come back. */
51
+ export function newTurnSignal(): TurnSignal {
52
+ return { finishReason: "", unparsedToolCallErrors: [], providerError: "", toolCallTurns: 0 };
53
+ }
54
+
55
+ /**
56
+ * Record one host completion frame into `signal`.
57
+ *
58
+ * Called by each runner's LLM bridge with the raw `llm_request` reply, so classification sees the
59
+ * host's own view of the turn rather than re-deriving it from pi's message log.
60
+ */
61
+ export function observeCompletion(signal: TurnSignal, reply: Record<string, any>): void {
62
+ if (reply.error) {
63
+ signal.providerError = String(reply.error);
64
+ return;
65
+ }
66
+ const choice = reply.completion?.choices?.[0] ?? {};
67
+ const message = choice.message ?? {};
68
+ signal.providerError = "";
69
+ signal.finishReason = String(choice.finish_reason ?? "stop");
70
+ const unparsed = choice.wmo_unparsed_tool_calls;
71
+ signal.unparsedToolCallErrors = Array.isArray(unparsed) ? unparsed.map((e: unknown) => String(e)) : [];
72
+ if (Array.isArray(message.tool_calls) && message.tool_calls.length > 0) {
73
+ signal.toolCallTurns += 1;
74
+ }
75
+ }
76
+
77
+ /** Runner-owned flags `classifyEnd` needs beyond the last completion. */
78
+ export interface EndContext {
79
+ /** The runner aborted the agent because the turn cap was reached. */
80
+ hitTurnCap: boolean;
81
+ /** pi's own terminal error message, if any (`agent.state.errorMessage`). */
82
+ agentError: string;
83
+ }
84
+
85
+ /**
86
+ * Why `agent.prompt()` returned, for an episode where `submit` was NOT called.
87
+ *
88
+ * Order matters: a dead provider explains everything downstream of it, an explicit parse error is
89
+ * more specific than "no tool call", and truncation at the cap is more specific still.
90
+ */
91
+ export function classifyEnd(signal: TurnSignal, context: EndContext): DoneReason {
92
+ if (signal.providerError) return "provider_error";
93
+ if (signal.unparsedToolCallErrors.length > 0) return "unparsed_tool_call";
94
+ if (signal.finishReason === "length") return "output_truncated";
95
+ if (context.hitTurnCap) return "max_turns";
96
+ if (context.agentError) return "provider_error";
97
+ return "no_tool_call";
98
+ }
99
+
100
+ /**
101
+ * Whether the runner should nudge instead of reporting `done`.
102
+ *
103
+ * A provider that is failing every call and a turn cap that has already fired are terminal: more
104
+ * prompts would only re-pay for the same failure. Everything else gets up to MAX_NONACTION_TURNS
105
+ * consecutive attempts to act or submit.
106
+ */
107
+ export function shouldNudge(
108
+ reason: DoneReason,
109
+ consecutiveNonActionTurns: number,
110
+ turns: number,
111
+ maxTurns: number,
112
+ ): boolean {
113
+ if (reason === "provider_error" || reason === "max_turns") return false;
114
+ if (consecutiveNonActionTurns >= MAX_NONACTION_TURNS) return false;
115
+ return turns < maxTurns;
116
+ }
117
+
118
+ const ACT_OR_SUBMIT =
119
+ "Continue the task: either call exactly one tool now, or call `submit` with your final answer " +
120
+ "if the task is already complete. Do not reply with prose alone.";
121
+
122
+ /** The observation fed back to the model in place of ending the episode. */
123
+ export function nudgeFor(reason: DoneReason, signal: TurnSignal, maxOutputTokens: number): string {
124
+ if (reason === "output_truncated") {
125
+ return (
126
+ `[ERROR] NONE of the actions you just requested were performed: your reply exceeded ` +
127
+ `${maxOutputTokens} output tokens and was cut off mid-emission. Re-issue the request, ` +
128
+ `breaking it into chunks each of which is well under ${maxOutputTokens} tokens. ` +
129
+ ACT_OR_SUBMIT
130
+ );
131
+ }
132
+ if (reason === "unparsed_tool_call") {
133
+ const warnings = signal.unparsedToolCallErrors.join("; ");
134
+ return (
135
+ `[ERROR] your tool call could not be parsed, so NOTHING was executed. Parser ` +
136
+ `warnings from your last reply: ${warnings}. Emit the call again in exactly the ` +
137
+ `documented format, closing every block you open. ` +
138
+ ACT_OR_SUBMIT
139
+ );
140
+ }
141
+ return `[ERROR] your last reply contained no tool call, so nothing happened. ${ACT_OR_SUBMIT}`;
142
+ }
@@ -0,0 +1,262 @@
1
+ """Run the vendored pi live-session peer as a local Node.js process.
2
+
3
+ The platform remains the credential boundary: the Node peer receives worker
4
+ completions over the existing stdio frame protocol and never receives provider
5
+ keys. Unlike the E2B backend, this module deliberately runs the harness process
6
+ on the user's machine. The CLI presents an explicit consent prompt before it
7
+ reaches this boundary.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import base64
13
+ import contextlib
14
+ import json
15
+ import os
16
+ import queue
17
+ import re
18
+ import shutil
19
+ import subprocess
20
+ import tempfile
21
+ import threading
22
+ from collections import deque
23
+ from pathlib import Path
24
+ from typing import TYPE_CHECKING, Protocol, TextIO, cast
25
+
26
+ from wmo.core.types import JsonObject
27
+ from wmo.harness.pi_e2b import (
28
+ HELLO_TIMEOUT_S,
29
+ PI_NPM_PACKAGES,
30
+ TRANSPORT_KEEPALIVE_TYPE,
31
+ session_entry_files,
32
+ )
33
+
34
+ if TYPE_CHECKING:
35
+ from collections.abc import Callable
36
+
37
+ _PI_VERSION = "0.80.3"
38
+ _MIN_NODE = (22, 19, 0)
39
+ _PACKAGE_JSON = '{"name":"wmo-pi-local","private":true,"type":"module"}\n'
40
+ _INSTALL_MARKER = ".wmo-pi-dependencies"
41
+ _STDERR_LINES = 50
42
+
43
+
44
+ class _CompletedCommand(Protocol):
45
+ """The subprocess result slice runtime bootstrap consumes."""
46
+
47
+ stdout: str
48
+
49
+
50
+ class _TextProcess(Protocol):
51
+ """The text-mode Popen slice used by the local frame channel."""
52
+
53
+ stdin: TextIO | None
54
+ stdout: TextIO | None
55
+ stderr: TextIO | None
56
+
57
+ def poll(self) -> int | None: ...
58
+
59
+ def wait(self, timeout: float | None = None) -> int: ...
60
+
61
+ def terminate(self) -> None: ...
62
+
63
+ def kill(self) -> None: ...
64
+
65
+
66
+ class _Eof:
67
+ """Reader-thread sentinel for a closed runner stdout stream."""
68
+
69
+
70
+ _EOF = _Eof()
71
+
72
+
73
+ def parse_node_version(output: str) -> tuple[int, int, int]:
74
+ """Parse ``node --version`` output into a semantic-version triple."""
75
+ match = re.fullmatch(r"v(\d+)\.(\d+)\.(\d+)\s*", output)
76
+ if match is None:
77
+ raise RuntimeError(f"could not parse Node.js version output: {output.strip()!r}")
78
+ major, minor, patch = match.groups()
79
+ return int(major), int(minor), int(patch)
80
+
81
+
82
+ def default_local_pi_runtime_dir() -> Path:
83
+ """Return the user cache directory for the pinned local pi runtime."""
84
+ cache = Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache"))
85
+ return cache / "wmo" / "pi" / _PI_VERSION
86
+
87
+
88
+ def ensure_local_pi_runtime(
89
+ runtime_dir: Path,
90
+ *,
91
+ node: str,
92
+ npm: str,
93
+ run_command: Callable[..., _CompletedCommand] = subprocess.run,
94
+ ) -> Path:
95
+ """Refresh the live runner and install its pinned npm dependencies once."""
96
+ version = run_command(
97
+ [node, "--version"],
98
+ capture_output=True,
99
+ text=True,
100
+ check=True,
101
+ ).stdout
102
+ parsed = parse_node_version(version)
103
+ if parsed < _MIN_NODE:
104
+ required = ".".join(str(part) for part in _MIN_NODE)
105
+ found = ".".join(str(part) for part in parsed)
106
+ raise RuntimeError(f"local pi requires Node.js {required} or newer (found {found})")
107
+
108
+ runtime_dir.mkdir(parents=True, exist_ok=True)
109
+ (runtime_dir / "package.json").write_text(_PACKAGE_JSON, encoding="utf-8")
110
+ for name, content in session_entry_files().items():
111
+ (runtime_dir / name).write_text(content, encoding="utf-8")
112
+
113
+ marker = runtime_dir / _INSTALL_MARKER
114
+ expected = "\n".join(PI_NPM_PACKAGES) + "\n"
115
+ if marker.is_file() and marker.read_text(encoding="utf-8") == expected:
116
+ return runtime_dir
117
+ run_command(
118
+ [npm, "install", "--no-audit", "--no-fund", "--ignore-scripts", *PI_NPM_PACKAGES],
119
+ cwd=runtime_dir,
120
+ check=True,
121
+ capture_output=True,
122
+ text=True,
123
+ timeout=600,
124
+ )
125
+ marker.write_text(expected, encoding="utf-8")
126
+ return runtime_dir
127
+
128
+
129
+ class LocalStdioChannel:
130
+ """A live-session frame channel over a local Node child process."""
131
+
132
+ def __init__(
133
+ self,
134
+ process: _TextProcess,
135
+ *,
136
+ stderr_lines: int = _STDERR_LINES,
137
+ cleanup_dir: Path | None = None,
138
+ ) -> None:
139
+ """Start bounded stdout/stderr reader threads for ``process``."""
140
+ if process.stdin is None or process.stdout is None or process.stderr is None:
141
+ raise RuntimeError("local pi process must expose stdin, stdout, and stderr pipes")
142
+ self._process = process
143
+ self._stdin = process.stdin
144
+ self._stdout = process.stdout
145
+ self._stderr_stream = process.stderr
146
+ self._frames: queue.Queue[JsonObject | _Eof] = queue.Queue()
147
+ self._stderr: deque[str] = deque(maxlen=stderr_lines)
148
+ self._cleanup_dir = cleanup_dir
149
+ self._closed = False
150
+ threading.Thread(target=self._read_stdout, name="pi-local-stdout", daemon=True).start()
151
+ threading.Thread(target=self._read_stderr, name="pi-local-stderr", daemon=True).start()
152
+
153
+ def send(self, frame: JsonObject) -> None:
154
+ """Write one base64(JSON) frame to the child process."""
155
+ line = base64.b64encode(json.dumps(frame).encode()).decode() + "\n"
156
+ self._stdin.write(line)
157
+ self._stdin.flush()
158
+
159
+ def recv(self, timeout: float | None = None) -> JsonObject | None:
160
+ """Return the next decoded frame, optionally bounded by ``timeout``."""
161
+ try:
162
+ item = self._frames.get(timeout=timeout)
163
+ except queue.Empty:
164
+ message = f"no frame from local pi within {timeout}s{self._stderr_suffix()}"
165
+ raise TimeoutError(message) from None
166
+ if isinstance(item, _Eof):
167
+ self._frames.put(item)
168
+ if self._closed:
169
+ return None
170
+ raise RuntimeError(f"local pi process exited unexpectedly{self._stderr_suffix()}")
171
+ return item
172
+
173
+ def close(self) -> None:
174
+ """Shut down the child process without leaving a local runner behind."""
175
+ if self._closed:
176
+ return
177
+ self._closed = True
178
+ with contextlib.suppress(Exception):
179
+ self.send({"type": "shutdown"})
180
+ with contextlib.suppress(Exception):
181
+ self._process.wait(timeout=2)
182
+ if self._process.poll() is None:
183
+ with contextlib.suppress(Exception):
184
+ self._process.terminate()
185
+ with contextlib.suppress(Exception):
186
+ self._process.wait(timeout=2)
187
+ if self._process.poll() is None:
188
+ with contextlib.suppress(Exception):
189
+ self._process.kill()
190
+ if self._cleanup_dir is not None:
191
+ with contextlib.suppress(OSError):
192
+ shutil.rmtree(self._cleanup_dir)
193
+
194
+ def _read_stdout(self) -> None:
195
+ try:
196
+ for raw in self._stdout:
197
+ text = raw.strip()
198
+ if not text:
199
+ continue
200
+ try:
201
+ frame = json.loads(base64.b64decode(text, validate=True))
202
+ except ValueError:
203
+ self._stderr.append(f"[stdout] {text}")
204
+ continue
205
+ if isinstance(frame, dict):
206
+ if frame.get("type") == TRANSPORT_KEEPALIVE_TYPE:
207
+ continue
208
+ self._frames.put(cast("JsonObject", frame))
209
+ else:
210
+ self._stderr.append(f"[stdout] {text}")
211
+ finally:
212
+ self._frames.put(_EOF)
213
+
214
+ def _read_stderr(self) -> None:
215
+ for raw in self._stderr_stream:
216
+ text = raw.rstrip()
217
+ if text:
218
+ self._stderr.append(text)
219
+
220
+ def _stderr_suffix(self) -> str:
221
+ tail = "\n".join(self._stderr)
222
+ return f"; recent stderr:\n{tail}" if tail else ""
223
+
224
+
225
+ def start_local_live_runner(
226
+ *,
227
+ runtime_dir: Path | None = None,
228
+ hello_timeout: float = HELLO_TIMEOUT_S,
229
+ ) -> LocalStdioChannel:
230
+ """Bootstrap and start the local pi peer, returning a hello-verified channel."""
231
+ node = shutil.which("node")
232
+ npm = shutil.which("npm")
233
+ if node is None or npm is None:
234
+ raise RuntimeError("local pi requires Node.js 22.19+ and npm on PATH")
235
+ root = ensure_local_pi_runtime(
236
+ runtime_dir or default_local_pi_runtime_dir(), node=node, npm=npm
237
+ )
238
+ # Each process gets a private cwd for materialized champion code. Node still
239
+ # resolves dependencies from the cached runner's parent directory.
240
+ process_root = Path(tempfile.mkdtemp(prefix="run-", dir=root))
241
+ try:
242
+ process = subprocess.Popen( # noqa: S603 - fixed executable/args, no shell
243
+ [node, "--experimental-strip-types", str(root / "runner_live.ts")],
244
+ cwd=process_root,
245
+ stdin=subprocess.PIPE,
246
+ stdout=subprocess.PIPE,
247
+ stderr=subprocess.PIPE,
248
+ text=True,
249
+ bufsize=1,
250
+ )
251
+ except BaseException:
252
+ shutil.rmtree(process_root, ignore_errors=True)
253
+ raise
254
+ channel = LocalStdioChannel(cast("_TextProcess", process), cleanup_dir=process_root)
255
+ try:
256
+ frame = channel.recv(timeout=hello_timeout)
257
+ if frame is None or frame.get("type") != "hello":
258
+ raise RuntimeError("local pi did not send its hello frame")
259
+ except BaseException:
260
+ channel.close()
261
+ raise
262
+ return channel