world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,488 @@
1
+ # @earendil-works/pi-agent-core
2
+
3
+ Stateful agent with tool execution and event streaming. Built on `@earendil-works/pi-ai`.
4
+
5
+ ## Installation
6
+
7
+ ```bash
8
+ npm install @earendil-works/pi-agent-core
9
+ ```
10
+
11
+ ## Quick Start
12
+
13
+ ```typescript
14
+ import { Agent } from "@earendil-works/pi-agent-core";
15
+ import { getModel } from "@earendil-works/pi-ai";
16
+
17
+ const agent = new Agent({
18
+ initialState: {
19
+ systemPrompt: "You are a helpful assistant.",
20
+ model: getModel("anthropic", "claude-sonnet-4-20250514"),
21
+ },
22
+ });
23
+
24
+ agent.subscribe((event) => {
25
+ if (event.type === "message_update" && event.assistantMessageEvent.type === "text_delta") {
26
+ // Stream just the new text chunk
27
+ process.stdout.write(event.assistantMessageEvent.delta);
28
+ }
29
+ });
30
+
31
+ await agent.prompt("Hello!");
32
+ ```
33
+
34
+ ## Core Concepts
35
+
36
+ ### AgentMessage vs LLM Message
37
+
38
+ The agent works with `AgentMessage`, a flexible type that can include:
39
+ - Standard LLM messages (`user`, `assistant`, `toolResult`)
40
+ - Custom app-specific message types via declaration merging
41
+
42
+ LLMs only understand `user`, `assistant`, and `toolResult`. The `convertToLlm` function bridges this gap by filtering and transforming messages before each LLM call.
43
+
44
+ ### Message Flow
45
+
46
+ ```
47
+ AgentMessage[] → transformContext() → AgentMessage[] → convertToLlm() → Message[] → LLM
48
+ (optional) (required)
49
+ ```
50
+
51
+ 1. **transformContext**: Prune old messages, inject external context
52
+ 2. **convertToLlm**: Filter out UI-only messages, convert custom types to LLM format
53
+
54
+ ## Event Flow
55
+
56
+ The agent emits events for UI updates. Understanding the event sequence helps build responsive interfaces.
57
+
58
+ ### prompt() Event Sequence
59
+
60
+ When you call `prompt("Hello")`:
61
+
62
+ ```
63
+ prompt("Hello")
64
+ ├─ agent_start
65
+ ├─ turn_start
66
+ ├─ message_start { message: userMessage } // Your prompt
67
+ ├─ message_end { message: userMessage }
68
+ ├─ message_start { message: assistantMessage } // LLM starts responding
69
+ ├─ message_update { message: partial... } // Streaming chunks
70
+ ├─ message_update { message: partial... }
71
+ ├─ message_end { message: assistantMessage } // Complete response
72
+ ├─ turn_end { message, toolResults: [] }
73
+ └─ agent_end { messages: [...] }
74
+ ```
75
+
76
+ ### With Tool Calls
77
+
78
+ If the assistant calls tools, the loop continues:
79
+
80
+ ```
81
+ prompt("Read config.json")
82
+ ├─ agent_start
83
+ ├─ turn_start
84
+ ├─ message_start/end { userMessage }
85
+ ├─ message_start { assistantMessage with toolCall }
86
+ ├─ message_update...
87
+ ├─ message_end { assistantMessage }
88
+ ├─ tool_execution_start { toolCallId, toolName, args }
89
+ ├─ tool_execution_update { partialResult } // If tool streams
90
+ ├─ tool_execution_end { toolCallId, result }
91
+ ├─ message_start/end { toolResultMessage }
92
+ ├─ turn_end { message, toolResults: [toolResult] }
93
+
94
+ ├─ turn_start // Next turn
95
+ ├─ message_start { assistantMessage } // LLM responds to tool result
96
+ ├─ message_update...
97
+ ├─ message_end
98
+ ├─ turn_end
99
+ └─ agent_end
100
+ ```
101
+
102
+ Tool execution mode is configurable:
103
+
104
+ - `parallel` (default): preflight tool calls sequentially, execute allowed tools concurrently, emit `tool_execution_end` as soon as each tool is finalized, then emit toolResult messages and `turn_end.toolResults` in assistant source order
105
+ - `sequential`: execute tool calls one by one, matching the historical behavior
106
+
107
+ In parallel mode, tool completion events follow tool completion order, but persisted toolResult messages still follow assistant source order.
108
+
109
+ The mode can be set globally via `toolExecution` in the agent config, or per-tool via `executionMode` on `AgentTool`. If any tool call in a batch targets a tool with `executionMode: "sequential"`, the entire batch executes sequentially regardless of the global setting.
110
+
111
+ The `beforeToolCall` hook runs after `tool_execution_start` and validated argument parsing. It can block execution. The `afterToolCall` hook runs after tool execution finishes and before `tool_execution_end` and final tool result message events are emitted.
112
+
113
+ Tools can also return `terminate: true` to hint that the automatic follow-up LLM call should be skipped. The loop only stops early when every finalized tool result in that batch sets `terminate: true`. Mixed batches continue normally.
114
+
115
+ Low-level loop callers can set `shouldStopAfterTurn` to stop gracefully after the current turn completes:
116
+
117
+ ```typescript
118
+ const stream = agentLoop(prompts, context, {
119
+ model,
120
+ convertToLlm,
121
+ shouldStopAfterTurn: async ({ message, toolResults, context, newMessages }) => {
122
+ return shouldCompactBeforeNextTurn(context.messages);
123
+ },
124
+ });
125
+ ```
126
+
127
+ `shouldStopAfterTurn` runs after `turn_end` is emitted and after the assistant response and any tool executions have completed normally. If it returns `true`, the loop emits `agent_end` and exits before polling steering or follow-up queues, and before starting another LLM call. It does not abort the provider stream, does not cancel running tools, and does not alter the assistant message stop reason.
128
+
129
+ When you use the `Agent` class, assistant `message_end` processing is treated as a barrier before tool preflight begins. That means `beforeToolCall` sees agent state that already includes the assistant message that requested the tool call.
130
+
131
+ ### continue() Event Sequence
132
+
133
+ `continue()` resumes from existing context without adding a new message. Use it for retries after errors.
134
+
135
+ ```typescript
136
+ // After an error, retry from current state
137
+ await agent.continue();
138
+ ```
139
+
140
+ The last message in context must be `user` or `toolResult` (not `assistant`).
141
+
142
+ ### Event Types
143
+
144
+ | Event | Description |
145
+ |-------|-------------|
146
+ | `agent_start` | Agent begins processing |
147
+ | `agent_end` | Final event for the run. Awaited subscribers for this event still count toward settlement |
148
+ | `turn_start` | New turn begins (one LLM call + tool executions) |
149
+ | `turn_end` | Turn completes with assistant message and tool results |
150
+ | `message_start` | Any message begins (user, assistant, toolResult) |
151
+ | `message_update` | **Assistant only.** Includes `assistantMessageEvent` with delta |
152
+ | `message_end` | Message completes |
153
+ | `tool_execution_start` | Tool begins |
154
+ | `tool_execution_update` | Tool streams progress |
155
+ | `tool_execution_end` | Tool completes |
156
+
157
+ `Agent.subscribe()` listeners are awaited in registration order. `agent_end` means no more loop events will be emitted, but `await agent.waitForIdle()` and `await agent.prompt(...)` only settle after awaited `agent_end` listeners finish.
158
+
159
+ ## Agent Options
160
+
161
+ ```typescript
162
+ const agent = new Agent({
163
+ // Initial state
164
+ initialState: {
165
+ systemPrompt: string,
166
+ model: Model<any>,
167
+ thinkingLevel: "off" | "minimal" | "low" | "medium" | "high" | "xhigh",
168
+ tools: AgentTool<any>[],
169
+ messages: AgentMessage[],
170
+ },
171
+
172
+ // Convert AgentMessage[] to LLM Message[] (required for custom message types)
173
+ convertToLlm: (messages) => messages.filter(...),
174
+
175
+ // Transform context before convertToLlm (for pruning, compaction)
176
+ transformContext: async (messages, signal) => pruneOldMessages(messages),
177
+
178
+ // Steering mode: "one-at-a-time" (default) or "all"
179
+ steeringMode: "one-at-a-time",
180
+
181
+ // Follow-up mode: "one-at-a-time" (default) or "all"
182
+ followUpMode: "one-at-a-time",
183
+
184
+ // Custom stream function (for proxy backends)
185
+ streamFn: streamProxy,
186
+
187
+ // Session ID for provider caching
188
+ sessionId: "session-123",
189
+
190
+ // Dynamic API key resolution (for expiring OAuth tokens)
191
+ getApiKey: async (provider) => refreshToken(),
192
+
193
+ // Tool execution mode: "parallel" (default) or "sequential"
194
+ toolExecution: "parallel",
195
+
196
+ // Preflight each tool call after args are validated. Can block execution.
197
+ beforeToolCall: async ({ toolCall, args, context }) => {
198
+ if (toolCall.name === "bash") {
199
+ return { block: true, reason: "bash is disabled" };
200
+ }
201
+ },
202
+
203
+ // Postprocess each tool result before final tool events are emitted.
204
+ afterToolCall: async ({ toolCall, result, isError, context }) => {
205
+ if (toolCall.name === "notify_done" && !isError) {
206
+ return { terminate: true };
207
+ }
208
+ if (!isError) {
209
+ return { details: { ...result.details, audited: true } };
210
+ }
211
+ },
212
+
213
+ // Custom thinking budgets for token-based providers
214
+ thinkingBudgets: {
215
+ minimal: 128,
216
+ low: 512,
217
+ medium: 1024,
218
+ high: 2048,
219
+ },
220
+ });
221
+ ```
222
+
223
+ ## Agent State
224
+
225
+ ```typescript
226
+ interface AgentState {
227
+ systemPrompt: string;
228
+ model: Model<any>;
229
+ thinkingLevel: ThinkingLevel;
230
+ tools: AgentTool<any>[];
231
+ messages: AgentMessage[];
232
+ readonly isStreaming: boolean;
233
+ readonly streamingMessage?: AgentMessage;
234
+ readonly pendingToolCalls: ReadonlySet<string>;
235
+ readonly errorMessage?: string;
236
+ }
237
+ ```
238
+
239
+ Access state via `agent.state`.
240
+
241
+ Assigning `agent.state.tools = [...]` or `agent.state.messages = [...]` copies the top-level array before storing it. Mutating the returned array mutates the current agent state.
242
+
243
+ During streaming, `agent.state.streamingMessage` contains the current partial assistant message.
244
+
245
+ `agent.state.isStreaming` remains `true` until the run fully settles, including awaited `agent_end` subscribers.
246
+
247
+ ## Methods
248
+
249
+ ### Prompting
250
+
251
+ ```typescript
252
+ // Text prompt
253
+ await agent.prompt("Hello");
254
+
255
+ // With images
256
+ await agent.prompt("What's in this image?", [
257
+ { type: "image", data: base64Data, mimeType: "image/jpeg" }
258
+ ]);
259
+
260
+ // AgentMessage directly
261
+ await agent.prompt({ role: "user", content: "Hello", timestamp: Date.now() });
262
+
263
+ // Continue from current context (last message must be user or toolResult)
264
+ await agent.continue();
265
+ ```
266
+
267
+ ### State Management
268
+
269
+ ```typescript
270
+ agent.state.systemPrompt = "New prompt";
271
+ agent.state.model = getModel("openai", "gpt-4o");
272
+ agent.state.thinkingLevel = "medium";
273
+ agent.state.tools = [myTool];
274
+ agent.toolExecution = "sequential";
275
+ agent.beforeToolCall = async ({ toolCall }) => undefined;
276
+ agent.afterToolCall = async ({ toolCall, result }) => undefined;
277
+ agent.state.messages = newMessages; // top-level array is copied
278
+ agent.state.messages.push(message);
279
+ agent.reset();
280
+ ```
281
+
282
+ ### Session and Thinking Budgets
283
+
284
+ ```typescript
285
+ agent.sessionId = "session-123";
286
+
287
+ agent.thinkingBudgets = {
288
+ minimal: 128,
289
+ low: 512,
290
+ medium: 1024,
291
+ high: 2048,
292
+ };
293
+ ```
294
+
295
+ ### Control
296
+
297
+ ```typescript
298
+ agent.abort(); // Cancel current operation
299
+ await agent.waitForIdle(); // Wait for completion
300
+ ```
301
+
302
+ ### Events
303
+
304
+ ```typescript
305
+ const unsubscribe = agent.subscribe(async (event, signal) => {
306
+ if (event.type === "agent_end") {
307
+ // Final barrier work for the run
308
+ await flushSessionState(signal);
309
+ }
310
+ });
311
+ unsubscribe();
312
+ ```
313
+
314
+ ## Steering and Follow-up
315
+
316
+ Steering messages let you interrupt the agent while tools are running. Follow-up messages let you queue work after the agent would otherwise stop.
317
+
318
+ ```typescript
319
+ agent.steeringMode = "one-at-a-time";
320
+ agent.followUpMode = "one-at-a-time";
321
+
322
+ // While agent is running tools
323
+ agent.steer({
324
+ role: "user",
325
+ content: "Stop! Do this instead.",
326
+ timestamp: Date.now(),
327
+ });
328
+
329
+ // After the agent finishes its current work
330
+ agent.followUp({
331
+ role: "user",
332
+ content: "Also summarize the result.",
333
+ timestamp: Date.now(),
334
+ });
335
+
336
+ const steeringMode = agent.steeringMode;
337
+ const followUpMode = agent.followUpMode;
338
+
339
+ agent.clearSteeringQueue();
340
+ agent.clearFollowUpQueue();
341
+ agent.clearAllQueues();
342
+ ```
343
+
344
+ Use clearSteeringQueue, clearFollowUpQueue, or clearAllQueues to drop queued messages.
345
+
346
+ When steering messages are detected after a turn completes:
347
+ 1. All tool calls from the current assistant message have already finished
348
+ 2. Steering messages are injected
349
+ 3. The LLM responds on the next turn
350
+
351
+ Follow-up messages are checked only when there are no more tool calls and no steering messages. If any are queued, they are injected and another turn runs.
352
+
353
+ ## Custom Message Types
354
+
355
+ Extend `AgentMessage` via declaration merging:
356
+
357
+ ```typescript
358
+ declare module "@earendil-works/pi-agent-core" {
359
+ interface CustomAgentMessages {
360
+ notification: { role: "notification"; text: string; timestamp: number };
361
+ }
362
+ }
363
+
364
+ // Now valid
365
+ const msg: AgentMessage = { role: "notification", text: "Info", timestamp: Date.now() };
366
+ ```
367
+
368
+ Handle custom types in `convertToLlm`:
369
+
370
+ ```typescript
371
+ const agent = new Agent({
372
+ convertToLlm: (messages) => messages.flatMap(m => {
373
+ if (m.role === "notification") return []; // Filter out
374
+ return [m];
375
+ }),
376
+ });
377
+ ```
378
+
379
+ ## Tools
380
+
381
+ Define tools using `AgentTool`:
382
+
383
+ ```typescript
384
+ import { Type } from "typebox";
385
+
386
+ const readFileTool: AgentTool = {
387
+ name: "read_file",
388
+ label: "Read File", // For UI display
389
+ description: "Read a file's contents",
390
+ parameters: Type.Object({
391
+ path: Type.String({ description: "File path" }),
392
+ }),
393
+ // Override execution mode for this tool (optional).
394
+ // "sequential" forces the entire batch to run one at a time.
395
+ // "parallel" allows concurrent execution with other tool calls.
396
+ // If omitted, the global toolExecution config applies.
397
+ executionMode: "sequential",
398
+ execute: async (toolCallId, params, signal, onUpdate) => {
399
+ const content = await fs.readFile(params.path, "utf-8");
400
+
401
+ // Optional: stream progress
402
+ onUpdate?.({ content: [{ type: "text", text: "Reading..." }], details: {} });
403
+
404
+ // Optional: add `terminate: true` here to skip the automatic follow-up LLM call
405
+ // when every finalized tool result in the batch does the same.
406
+ return {
407
+ content: [{ type: "text", text: content }],
408
+ details: { path: params.path, size: content.length },
409
+ };
410
+ },
411
+ };
412
+
413
+ agent.state.tools = [readFileTool];
414
+ ```
415
+
416
+ ### Error Handling
417
+
418
+ **Throw an error** when a tool fails. Do not return error messages as content.
419
+
420
+ ```typescript
421
+ execute: async (toolCallId, params, signal, onUpdate) => {
422
+ if (!fs.existsSync(params.path)) {
423
+ throw new Error(`File not found: ${params.path}`);
424
+ }
425
+ // Return content only on success
426
+ return { content: [{ type: "text", text: "..." }] };
427
+ }
428
+ ```
429
+
430
+ Thrown errors are caught by the agent and reported to the LLM as tool errors with `isError: true`.
431
+
432
+ Return `terminate: true` from `execute()` or `afterToolCall` to hint that the agent should stop after the current tool batch. This only takes effect when every finalized tool result in the batch is terminating. The hint is runtime-only; emitted `toolResult` transcript messages remain standard LLM tool results.
433
+
434
+ ## Proxy Usage
435
+
436
+ For browser apps that proxy through a backend:
437
+
438
+ ```typescript
439
+ import { Agent, streamProxy } from "@earendil-works/pi-agent-core";
440
+
441
+ const agent = new Agent({
442
+ streamFn: (model, context, options) =>
443
+ streamProxy(model, context, {
444
+ ...options,
445
+ authToken: "...",
446
+ proxyUrl: "https://your-server.com",
447
+ }),
448
+ });
449
+ ```
450
+
451
+ ## Low-Level API
452
+
453
+ For direct control without the Agent class:
454
+
455
+ ```typescript
456
+ import { agentLoop, agentLoopContinue } from "@earendil-works/pi-agent-core";
457
+
458
+ const context: AgentContext = {
459
+ systemPrompt: "You are helpful.",
460
+ messages: [],
461
+ tools: [],
462
+ };
463
+
464
+ const config: AgentLoopConfig = {
465
+ model: getModel("openai", "gpt-4o"),
466
+ convertToLlm: (msgs) => msgs.filter(m => ["user", "assistant", "toolResult"].includes(m.role)),
467
+ toolExecution: "parallel", // overridden by per-tool executionMode if set
468
+ beforeToolCall: async ({ toolCall, args, context }) => undefined,
469
+ afterToolCall: async ({ toolCall, result, isError, context }) => undefined,
470
+ };
471
+
472
+ const userMessage = { role: "user", content: "Hello", timestamp: Date.now() };
473
+
474
+ for await (const event of agentLoop([userMessage], context, config)) {
475
+ console.log(event.type);
476
+ }
477
+
478
+ // Continue from existing context
479
+ for await (const event of agentLoopContinue(context, config)) {
480
+ console.log(event.type);
481
+ }
482
+ ```
483
+
484
+ These low-level streams are observational. They preserve event order, but they do not wait for your async event handling to settle before later producer phases continue. If you need message processing to act as a barrier before tool preflight, use the `Agent` class instead of raw `agentLoop()` or `agentLoopContinue()`.
485
+
486
+ ## License
487
+
488
+ MIT
@@ -0,0 +1,39 @@
1
+ # Vendored: pi agent (`@earendil-works/pi-agent-core`)
2
+
3
+ This directory is a **byte-exact copy** of the `packages/agent` subtree of the pi agent, vendored
4
+ into this repo so the harness search runs the real agent's source rather than a re-implementation.
5
+
6
+ | | |
7
+ |---|---|
8
+ | Upstream | https://github.com/earendil-works/pi |
9
+ | Package | `@earendil-works/pi-agent-core` |
10
+ | Tag | `v0.80.3` |
11
+ | Commit (the pin) | `a23abe4a695df8b69b613f73e9fdda2a8af894d4` |
12
+ | Vendored subtree | `packages/agent/` → this directory (`wmo/harness/vendor/pi-agent/`) |
13
+ | Files | 56 (byte-identical to upstream `packages/agent`) |
14
+ | Upstream license | MIT (© 2025 Mario Zechner) |
15
+
16
+ The tag is a convenience label; **the commit SHA is the pin**. `wmo/harness/vendor/vendor_pi.sh` fetches upstream
17
+ at that SHA, re-materializes this tree, and regenerates `wmo/harness/vendor/manifest.sha256`
18
+ (the per-file integrity ledger). Re-running it must produce zero diff against the committed copy —
19
+ that is how anyone re-verifies this vendoring from scratch.
20
+
21
+ ## License
22
+
23
+ The `packages/agent` package carries no license file of its own upstream; pi is MIT-licensed at the
24
+ repository root, so `LICENSE` here is a verbatim copy of the upstream **root** `LICENSE` at the
25
+ pinned commit. See it for the full MIT text and copyright.
26
+
27
+ ## Do not edit in place
28
+
29
+ These files are upstream bytes and must stay that way — `wmo/harness/vendor/vendor_pi.sh` and the checksum gate
30
+ both assume byte-identity with upstream. The harness *searches over* this source by loading it into
31
+ `code:` surfaces (`wmo/harness/pi_vendor.py`) and mutating those surfaces through audited
32
+ `HarnessDelta`s; the mutations live in stored `HarnessDoc` versions, never as edits to this tree.
33
+
34
+ ## What consumes it
35
+
36
+ `wmo/harness/pi_vendor.py` is the only reader: `pi_agent_code_surfaces()` loads the 25 runnable
37
+ `src/**/*.ts` files (vitest `*.test.ts` specs excluded) into `code:` surfaces, which
38
+ `wmo/harness/pi_runtime.py` materializes to run pi headless against the world model. The full
39
+ package is vendored on disk; only `src/` is surfaced.