world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,268 @@
1
+ /**
2
+ * Headless pi-agent entrypoint driven by the Python world-model shim.
3
+ *
4
+ * Run: PI_SHIM_URL=http://127.0.0.1:$PORT node --experimental-strip-types entry.ts
5
+ *
6
+ * Flow:
7
+ * 1. GET $PI_SHIM_URL/task -> {instruction, system, tools[]}
8
+ * 2. Build a pi Agent whose Model.baseUrl = $PI_SHIM_URL + "/v1" and
9
+ * api = "openai-completions" (so streamSimple hits the shim's SSE endpoint).
10
+ * 3. Register each task tool as an AgentTool whose execute() POSTs /tool.
11
+ * 4. Register a `submit` tool whose execute() POSTs /done {answer, reason:"submit"} and
12
+ * terminates the loop (AgentToolResult.terminate = true).
13
+ * 5. agent.prompt(instruction); a turn WITHOUT tool calls is not a completion, so read the
14
+ * shim's GET /signal (the host's view of the last completion) and nudge, bounded, before
15
+ * POSTing /done with the classified reason. See runner_termination.ts.
16
+ *
17
+ * Lives next to src/ when materialized on the runner (import paths "./src/agent.ts",
18
+ * "./runner_termination.ts").
19
+ */
20
+ import { Agent } from "./src/agent.ts";
21
+ import type { AgentTool, AgentToolResult } from "./src/types.ts";
22
+ import type { Model } from "@earendil-works/pi-ai";
23
+ import {
24
+ classifyEnd,
25
+ newTurnSignal,
26
+ nudgeFor,
27
+ shouldNudge,
28
+ type DoneReason,
29
+ } from "./runner_termination.ts";
30
+
31
+ const SHIM = process.env.PI_SHIM_URL;
32
+ if (!SHIM) {
33
+ console.error("PI_SHIM_URL not set");
34
+ process.exit(2);
35
+ }
36
+ const BASE = SHIM.replace(/\/$/, "");
37
+ const DEFAULT_MAX_TURNS = 20;
38
+ // Last-resort model context window when /task carries none. The host resolves the REAL served
39
+ // window (provider/SDK model info) and reports it as context_window; never assume a size here.
40
+ const DEFAULT_CONTEXT_WINDOW = 128000;
41
+
42
+ interface TaskTool {
43
+ name: string;
44
+ description: string;
45
+ parameters: any;
46
+ }
47
+ interface Task {
48
+ instruction: string;
49
+ system?: string;
50
+ tools: TaskTool[];
51
+ max_turns?: number;
52
+ max_output_tokens?: number;
53
+ context_window?: number;
54
+ }
55
+ /** The shim's host-side view of the most recent worker completion. */
56
+ interface ShimSignal {
57
+ finish_reason?: string;
58
+ unparsed_tool_calls?: string[];
59
+ provider_error?: string;
60
+ tool_call_turns?: number;
61
+ }
62
+
63
+ async function getJson<T>(path: string): Promise<T> {
64
+ const res = await fetch(BASE + path);
65
+ if (!res.ok) throw new Error(`GET ${path} -> ${res.status}`);
66
+ return (await res.json()) as T;
67
+ }
68
+
69
+ async function postJson<T>(path: string, body: unknown): Promise<T> {
70
+ const res = await fetch(BASE + path, {
71
+ method: "POST",
72
+ headers: { "Content-Type": "application/json" },
73
+ body: JSON.stringify(body),
74
+ });
75
+ if (!res.ok) throw new Error(`POST ${path} -> ${res.status}`);
76
+ return (await res.json()) as T;
77
+ }
78
+
79
+ let doneSent = false;
80
+ async function sendDone(answer: string | null, reason: DoneReason): Promise<void> {
81
+ if (doneSent) return;
82
+ doneSent = true;
83
+ await postJson("/done", { answer, reason });
84
+ }
85
+
86
+ /** Pull the host's view of the last completion into the shared TurnSignal shape. */
87
+ async function readSignal(toolCallTurns: number): Promise<ReturnType<typeof newTurnSignal>> {
88
+ const signal = newTurnSignal();
89
+ signal.toolCallTurns = toolCallTurns;
90
+ try {
91
+ const shim = await getJson<ShimSignal>("/signal");
92
+ signal.finishReason = String(shim.finish_reason ?? "");
93
+ signal.unparsedToolCallErrors = Array.isArray(shim.unparsed_tool_calls)
94
+ ? shim.unparsed_tool_calls.map((e) => String(e))
95
+ : [];
96
+ signal.providerError = String(shim.provider_error ?? "");
97
+ signal.toolCallTurns = Number.isInteger(shim.tool_call_turns)
98
+ ? Number(shim.tool_call_turns)
99
+ : toolCallTurns;
100
+ } catch (e) {
101
+ // An older shim has no /signal endpoint. Classification then degrades to "no_tool_call",
102
+ // which is still honest (it is never reported as a submission).
103
+ console.error(`[entry] /signal unavailable: ${e}`);
104
+ }
105
+ return signal;
106
+ }
107
+
108
+ // pi often ends by writing its final answer as a normal assistant message rather than calling
109
+ // `submit`. Capture the latest assistant text so we can use it as the answer if the loop exits
110
+ // without a submit call (otherwise the answer would be empty).
111
+ let lastAssistantText = "";
112
+ function assistantText(msg: any): string {
113
+ if (!msg || msg.role !== "assistant" || !Array.isArray(msg.content)) return "";
114
+ return msg.content
115
+ .filter((c: any) => c?.type === "text")
116
+ .map((c: any) => String(c.text ?? ""))
117
+ .join("")
118
+ .trim();
119
+ }
120
+
121
+ function makeShimTool(t: TaskTool): AgentTool<any> {
122
+ return {
123
+ name: t.name,
124
+ label: t.name,
125
+ description: t.description,
126
+ parameters: t.parameters,
127
+ execute: async (_id, params): Promise<AgentToolResult<any>> => {
128
+ const r = await postJson<{ content: string; is_error?: boolean }>("/tool", {
129
+ name: t.name,
130
+ arguments: params,
131
+ });
132
+ return {
133
+ content: [{ type: "text", text: String(r.content ?? "") }],
134
+ details: r,
135
+ terminate: false,
136
+ };
137
+ },
138
+ };
139
+ }
140
+
141
+ function makeSubmitTool(): AgentTool<any> {
142
+ return {
143
+ name: "submit",
144
+ label: "submit",
145
+ description: "Submit the final answer and finish the task.",
146
+ parameters: {
147
+ type: "object",
148
+ properties: { answer: { type: "string" } },
149
+ required: ["answer"],
150
+ },
151
+ execute: async (_id, params: { answer: string }): Promise<AgentToolResult<any>> => {
152
+ await sendDone(params.answer ?? "", "submit");
153
+ return {
154
+ content: [{ type: "text", text: "submitted" }],
155
+ details: { answer: params.answer },
156
+ terminate: true, // stop the agent loop after this tool batch
157
+ };
158
+ },
159
+ };
160
+ }
161
+
162
+ async function main(): Promise<void> {
163
+ const task = await getJson<Task>("/task");
164
+ const configuredMaxTurns = task.max_turns;
165
+ const maxTurns =
166
+ configuredMaxTurns !== undefined &&
167
+ Number.isInteger(configuredMaxTurns) &&
168
+ configuredMaxTurns >= 1
169
+ ? configuredMaxTurns
170
+ : DEFAULT_MAX_TURNS;
171
+ const configuredMaxOutputTokens = task.max_output_tokens;
172
+ const maxOutputTokens =
173
+ configuredMaxOutputTokens !== undefined &&
174
+ Number.isInteger(configuredMaxOutputTokens) &&
175
+ configuredMaxOutputTokens >= 1
176
+ ? configuredMaxOutputTokens
177
+ : 4096;
178
+ const configuredContextWindow = task.context_window;
179
+ const contextWindow =
180
+ configuredContextWindow !== undefined &&
181
+ Number.isInteger(configuredContextWindow) &&
182
+ configuredContextWindow >= 1024
183
+ ? configuredContextWindow
184
+ : DEFAULT_CONTEXT_WINDOW;
185
+
186
+ const model: Model<"openai-completions"> = {
187
+ id: "stub-model",
188
+ name: "stub-model",
189
+ api: "openai-completions",
190
+ provider: "shim", // non-builtin provider -> uses model.baseUrl directly
191
+ baseUrl: BASE + "/v1",
192
+ reasoning: false,
193
+ input: ["text"],
194
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
195
+ contextWindow,
196
+ maxTokens: maxOutputTokens,
197
+ };
198
+
199
+ // `submit` is provided by entry.ts (it drives /done + loop termination); drop any
200
+ // task-supplied `submit` so the tool list pi sends the model has unique names.
201
+ const envTools = task.tools.filter((t) => t.name !== "submit");
202
+ const tools: AgentTool<any>[] = [...envTools.map(makeShimTool), makeSubmitTool()];
203
+
204
+ const agent = new Agent({
205
+ initialState: {
206
+ systemPrompt: task.system ?? "",
207
+ model,
208
+ tools,
209
+ },
210
+ // apiKey passed via stream options; shim ignores it but SDK requires non-empty.
211
+ getApiKey: () => "x",
212
+ });
213
+
214
+ // Hard turn cap: use the harness document's per-episode value.
215
+ let turnCount = 0;
216
+ let hitTurnCap = false;
217
+ agent.subscribe((event) => {
218
+ if (event.type === "turn_end" || event.type === "message_end") {
219
+ const t = assistantText((event as any).message);
220
+ if (t) lastAssistantText = t;
221
+ }
222
+ if (event.type === "turn_end") {
223
+ turnCount += 1;
224
+ if (turnCount >= maxTurns) {
225
+ hitTurnCap = true;
226
+ agent.abort();
227
+ }
228
+ }
229
+ });
230
+
231
+ await agent.prompt(task.instruction);
232
+
233
+ // A turn without tool calls is NOT a completion. Nudge (bounded) before ending, then report
234
+ // exactly why this episode stopped so the host never records it as a submission.
235
+ let signal = await readSignal(0);
236
+ let reason: DoneReason = classifyEnd(signal, {
237
+ hitTurnCap,
238
+ agentError: String(agent.state.errorMessage ?? ""),
239
+ });
240
+ let consecutiveNonAction = 1;
241
+ while (!doneSent && shouldNudge(reason, consecutiveNonAction, turnCount, maxTurns)) {
242
+ const before = signal.toolCallTurns;
243
+ await agent.prompt(nudgeFor(reason, signal, maxOutputTokens));
244
+ signal = await readSignal(before);
245
+ consecutiveNonAction = signal.toolCallTurns > before ? 1 : consecutiveNonAction + 1;
246
+ reason = classifyEnd(signal, {
247
+ hitTurnCap,
248
+ agentError: String(agent.state.errorMessage ?? ""),
249
+ });
250
+ }
251
+
252
+ // Ensure /done was sent. If pi never called submit, fall back to its last assistant message
253
+ // text (its de-facto answer) rather than reporting empty.
254
+ if (!doneSent) {
255
+ const err = agent.state.errorMessage;
256
+ await sendDone(err ? null : lastAssistantText, reason);
257
+ }
258
+
259
+ console.error(
260
+ `[entry] done sent=${doneSent} reason=${reason} turns=${turnCount} err=${agent.state.errorMessage ?? ""}`,
261
+ );
262
+ process.exit(0);
263
+ }
264
+
265
+ main().catch((e) => {
266
+ console.error("[entry] fatal", e);
267
+ process.exit(1);
268
+ });
@@ -0,0 +1,92 @@
1
+ /**
2
+ * Length-prefixed JSON frame client for the RunnerLink transport (the Node peer of
3
+ * wmo/harness/runner_link.py). One TCP socket to the host carries every episode; requests get a
4
+ * fresh `req_id` and resolve when the matching response frame arrives, while server-pushed frames
5
+ * (episode_start, cancel, ping) fire registered handlers. Wire format matches runner_link.py
6
+ * exactly: 4-byte big-endian length prefix + UTF-8 JSON body.
7
+ */
8
+ import net from "node:net";
9
+
10
+ export type Frame = Record<string, any>;
11
+
12
+ export class FrameConn {
13
+ private sock: net.Socket;
14
+ private buf: Buffer = Buffer.alloc(0);
15
+ private waiters = new Map<number, (f: Frame) => void>();
16
+ private handlers = new Map<string, (f: Frame) => void>();
17
+ private reqSeq = 0;
18
+ private closed = false;
19
+
20
+ constructor(sock: net.Socket) {
21
+ this.sock = sock;
22
+ sock.on("data", (d: Buffer) => this._onData(d));
23
+ // If the channel drops while a request is in flight, settle every pending waiter with an
24
+ // error frame — otherwise awaiting llm_request/tool_request promises hang forever and the
25
+ // episode never returns a done/episode_error.
26
+ sock.on("close", () => this._settleAll("runner channel closed"));
27
+ sock.on("error", (e: Error) => this._settleAll(`runner channel error: ${e.message}`));
28
+ }
29
+
30
+ private _settleAll(reason: string): void {
31
+ this.closed = true;
32
+ const pending = [...this.waiters.values()];
33
+ this.waiters.clear();
34
+ for (const resolve of pending) {
35
+ resolve({ error: reason, content: reason, is_error: true });
36
+ }
37
+ }
38
+
39
+ /** Register a handler for a server-pushed frame type (no req_id): episode_start, cancel, ping. */
40
+ on(type: string, handler: (f: Frame) => void): void {
41
+ this.handlers.set(type, handler);
42
+ }
43
+
44
+ send(frame: Frame): void {
45
+ const body = Buffer.from(JSON.stringify(frame), "utf8");
46
+ const hdr = Buffer.alloc(4);
47
+ hdr.writeUInt32BE(body.length, 0);
48
+ this.sock.write(Buffer.concat([hdr, body]));
49
+ }
50
+
51
+ /** Send a request frame with a fresh req_id; resolve with the matching response frame. */
52
+ request(type: string, payload: Frame): Promise<Frame> {
53
+ if (this.closed) {
54
+ const reason = "runner channel closed";
55
+ return Promise.resolve({ error: reason, content: reason, is_error: true });
56
+ }
57
+ const req_id = ++this.reqSeq;
58
+ return new Promise((resolve) => {
59
+ this.waiters.set(req_id, resolve);
60
+ this.send({ type, req_id, ...payload });
61
+ });
62
+ }
63
+
64
+ private _onData(chunk: Buffer): void {
65
+ this.buf = Buffer.concat([this.buf, chunk]);
66
+ while (this.buf.length >= 4) {
67
+ const n = this.buf.readUInt32BE(0);
68
+ if (this.buf.length < 4 + n) break;
69
+ const body = this.buf.subarray(4, 4 + n);
70
+ this.buf = this.buf.subarray(4 + n);
71
+ let frame: Frame;
72
+ try {
73
+ frame = JSON.parse(body.toString("utf8"));
74
+ } catch {
75
+ continue;
76
+ }
77
+ this._dispatch(frame);
78
+ }
79
+ }
80
+
81
+ private _dispatch(frame: Frame): void {
82
+ const rid = frame.req_id;
83
+ if (typeof rid === "number" && this.waiters.has(rid)) {
84
+ const resolve = this.waiters.get(rid);
85
+ this.waiters.delete(rid);
86
+ resolve?.(frame);
87
+ return;
88
+ }
89
+ const handler = this.handlers.get(frame.type);
90
+ if (handler) handler(frame);
91
+ }
92
+ }