world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,212 @@
1
+ # Durable AgentHarness and session design
2
+
3
+ <!-- Synced from jot zmnps2zu. Edit this file in-repo going forward. -->
4
+
5
+ Durable AgentHarness / session design notes.
6
+
7
+ ## Framing
8
+
9
+ A fully durable `AgentHarness` is not realistic by itself because important dependencies are runtime JS supplied by the host app:
10
+
11
+ - tool implementations
12
+ - model/auth providers
13
+ - extensions and hook handlers
14
+ - resource loaders
15
+ - system-prompt callbacks/modifiers
16
+
17
+ Tool registries are runtime dependencies. The harness should persist serializable tool configuration, such as active tool names, but not concrete tool implementations.
18
+
19
+ The practical target is a semi-durable harness:
20
+
21
+ - session is the durable append-only state tree
22
+ - harness persists the state it owns into session entries
23
+ - the host app is responsible for recreating compatible non-persistable dependencies on resume
24
+ - recovery restarts from durable boundaries, not from an in-flight provider stream
25
+
26
+ ## Session owns durable state
27
+
28
+ Treat session as all durable agent state, not just transcript history.
29
+
30
+ Existing session state already includes harness state:
31
+
32
+ - model changes
33
+ - thinking-level changes
34
+ - active-tool changes
35
+ - leaf entries
36
+ - labels
37
+ - compactions and branch summaries
38
+ - custom messages and custom entries
39
+
40
+ That suggests continuing with one durable session log rather than adding harness sidecars. Sidecars may still be useful for large blobs, but the session entry should remain the source-of-truth reference.
41
+
42
+ ## What the app must provide on resume
43
+
44
+ The app must recreate compatible runtime dependencies:
45
+
46
+ - model registry / model objects
47
+ - tool registry
48
+ - extension set, versions, and ordering
49
+ - resource loaders
50
+ - system prompt providers/hooks
51
+ - auth providers
52
+ - app-specific hooks
53
+
54
+ Harness can validate stable IDs/versions/hashes when available, but it cannot serialize these dependencies itself.
55
+
56
+ ## Runtime configuration and restore
57
+
58
+ Constructor options remain explicit runtime configuration and do not read session state. Hidden async restore in a constructor would make failure handling ambiguous.
59
+
60
+ A future async builder/factory should own durable restore:
61
+
62
+ ```ts
63
+ const harness = await AgentHarness.builder()
64
+ .env(env)
65
+ .session(session)
66
+ .model(defaultModel)
67
+ .tools(runtimeTools)
68
+ .defaultActiveTools(["read", "edit"])
69
+ .restore({ missingActiveTools: "fail" });
70
+ ```
71
+
72
+ `restore()` should read the active branch, reduce durable harness configuration, apply defaults for missing entries, validate against app-supplied runtime dependencies, construct the harness, and optionally emit `source: "restore"` update events after construction.
73
+
74
+ For active tools:
75
+
76
+ - `active_tools_change` entries are branch-scoped durable config.
77
+ - If no `active_tools_change` exists on the branch, restore uses builder defaults, or all registered tools if no default active names were supplied.
78
+ - Active tool names must be unique.
79
+ - Tool registry names must be unique.
80
+ - Missing restored active tool names should fail restore by default; permissive drop/disable policies can be added explicitly later.
81
+ - Concrete tools are never restored from session; the host app must provide compatible tools.
82
+
83
+ ## What harness should persist
84
+
85
+ Minimum useful durability entries:
86
+
87
+ - branch-scoped active tool names
88
+ - queued steer/followUp/nextTurn messages
89
+ - queue consumption tied to a turn
90
+ - pending session writes accepted during active operations
91
+ - pending write application status
92
+ - operation start/finish/interruption
93
+ - turn start/finish
94
+ - provider request start/finish, if needed for recovery diagnostics
95
+ - tool call start/finish, if we want safe tool recovery
96
+
97
+ Potential entries:
98
+
99
+ ```ts
100
+ type DurableHarnessEntry =
101
+ | QueueEnqueuedEntry
102
+ | QueueConsumedEntry
103
+ | PendingWriteEnqueuedEntry
104
+ | PendingWriteAppliedEntry
105
+ | OperationStartedEntry
106
+ | OperationFinishedEntry
107
+ | OperationInterruptedEntry
108
+ | TurnStartedEntry
109
+ | TurnFinishedEntry
110
+ | ProviderRequestStartedEntry
111
+ | ProviderRequestFinishedEntry
112
+ | ToolCallStartedEntry
113
+ | ToolCallFinishedEntry;
114
+ ```
115
+
116
+ Every accepted mutation must be durable before the public API resolves.
117
+
118
+ ## Recovery model
119
+
120
+ On startup:
121
+
122
+ 1. Host app registers tools/models/extensions/resources/auth/hooks.
123
+ 2. Harness opens session.
124
+ 3. Harness reduces session entries into:
125
+ - current leaf
126
+ - conversation branch
127
+ - harness config, including active tool names
128
+ - queues
129
+ - pending writes
130
+ - active operation/turn/tool state
131
+ 4. Harness validates required runtime dependencies, including restored active tool names against the app-provided tool registry.
132
+ 5. Harness reconciles unfinished operation state.
133
+
134
+ Provider streams are not resumable. Recovery can only retry from a durable boundary or mark the operation interrupted.
135
+
136
+ ## Recovery policies
137
+
138
+ Default conservative policy:
139
+
140
+ - unfinished agent turn: mark interrupted, preserve durable queues/pending writes, return idle
141
+ - unfinished provider request: mark interrupted; do not retry automatically
142
+ - unfinished tool call: append interrupted/error tool result; retry only if the tool declares retry-safe/idempotent
143
+ - unfinished compaction: rerun if no compaction entry exists
144
+ - unfinished branch summary/tree navigation: rerun/apply missing summary or leaf entries if safe
145
+
146
+ Optional policy:
147
+
148
+ ```ts
149
+ recovery: "mark_interrupted" | "retry_unfinished"
150
+ ```
151
+
152
+ `retry_unfinished` must be guarded around non-idempotent tool calls.
153
+
154
+ ## Critical scenarios
155
+
156
+ ### Queues
157
+
158
+ - Crash before `queue_enqueued`: message was not accepted.
159
+ - Crash after `queue_enqueued`: message is restored.
160
+ - Crash after queue drain but before durable turn record: risk of loss/duplication.
161
+ - Required invariant: consumed queue IDs must be recorded in `turn_started` or equivalent before they are considered consumed.
162
+
163
+ ### Pending writes
164
+
165
+ - Crash before `pending_write_enqueued`: write was not accepted.
166
+ - Crash after enqueue before apply: recovery applies it.
167
+ - Crash after apply before applied marker: deterministic target entry IDs let recovery detect the entry already exists and mark it applied.
168
+
169
+ ### Agent loop turn
170
+
171
+ - Crash before provider request: retry or mark interrupted.
172
+ - Crash during provider request: mark interrupted by default.
173
+ - Crash after provider response before assistant message persisted: response is lost unless provider result was journaled.
174
+ - Crash after assistant message persisted: recover from durable message.
175
+
176
+ ### Tool calls
177
+
178
+ - Crash after tool call starts but before result: external side effects may already have happened.
179
+ - Default recovery should not rerun non-idempotent tools.
180
+ - Tool calls need stable IDs and retry-safety metadata for automatic recovery.
181
+
182
+ ### Compaction
183
+
184
+ - Crash before summary generation: rerun preparation/summary.
185
+ - Crash after generated summary but before compaction entry: rerun unless summary was journaled.
186
+ - Crash after compaction entry: operation is complete; append finish marker if missing.
187
+
188
+ ### Branch summary / tree navigation
189
+
190
+ - Crash before summary: rerun or mark interrupted.
191
+ - Crash after summary entry before leaf entry: append missing leaf entry.
192
+ - Crash after leaf entry: operation is complete; append finish marker if missing.
193
+
194
+ ## Minimum viable spike
195
+
196
+ 1. Add durable queue entries.
197
+ 2. Add durable pending write entries with deterministic target IDs.
198
+ 3. Add operation start/finish/interrupted entries.
199
+ 4. Add turn start with consumed queue IDs.
200
+ 5. Recover by reducing the session log.
201
+ 6. Mark unfinished agent turns interrupted by default.
202
+ 7. Rerun unfinished compaction/tree operations only when no final entry exists.
203
+ 8. Do not retry unfinished tool calls unless tool metadata says retry-safe.
204
+
205
+ ## Open questions
206
+
207
+ - Which remaining harness config entries should move into session first: resources, stream options, system prompt refs?
208
+ - Should resolved system prompt text be snapshotted per turn for audit/debug?
209
+ - Do we require strict dependency ID/version matching on resume?
210
+ - How much provider request data should be journaled?
211
+ - Should recovery append user-visible assistant interruption messages or only internal operation entries?
212
+ - Should storage support truncating a final partial JSONL line during recovery?
@@ -0,0 +1,445 @@
1
+ # AgentHarness hooks design
2
+
3
+ <!-- Synced from jot 3utlzkxy. Edit this file in-repo going forward. -->
4
+
5
+ Final design.
6
+
7
+ ## Core model
8
+
9
+ Events carry their result type as a type-only phantom:
10
+
11
+ ```ts
12
+ declare const HookResult: unique symbol;
13
+
14
+ interface HookEvent<TType extends string, TResult = void> {
15
+ type: TType;
16
+ readonly [HookResult]?: TResult;
17
+ }
18
+
19
+ type ResultOf<E> = E extends { readonly [HookResult]?: infer R } ? R : void;
20
+
21
+ type HookHandler<E, Ctx> = (
22
+ event: E,
23
+ ctx: Ctx,
24
+ signal?: AbortSignal,
25
+ ) => ResultOf<E> | void | Promise<ResultOf<E> | void>;
26
+
27
+ type HookObserver<E, Ctx> = (
28
+ event: E,
29
+ ctx: Ctx,
30
+ signal?: AbortSignal,
31
+ ) => void | Promise<void>;
32
+ ```
33
+
34
+ Example:
35
+
36
+ ```ts
37
+ interface ContextEvent extends HookEvent<"context", { messages?: AgentMessage[] }> {
38
+ type: "context";
39
+ messages: AgentMessage[];
40
+ }
41
+
42
+ interface ToolCallEvent extends HookEvent<"tool_call", { block?: boolean; reason?: string }> {
43
+ type: "tool_call";
44
+ toolName: string;
45
+ input: Record<string, unknown>;
46
+ }
47
+
48
+ interface MessageEndEvent extends HookEvent<"message_end"> {
49
+ type: "message_end";
50
+ message: AgentMessage;
51
+ }
52
+ ```
53
+
54
+ No result map. No spec table. The event type defines its own result.
55
+
56
+ ## Hooks interface
57
+
58
+ ```ts
59
+ interface AgentHarnessHooks<E extends HookEvent<string, unknown>, Ctx> {
60
+ context: Ctx;
61
+
62
+ setContext(ctx: Ctx): void;
63
+
64
+ observe(handler: HookObserver<E, Ctx>): () => void;
65
+
66
+ on<TType extends E["type"]>(
67
+ type: TType,
68
+ handler: HookHandler<Extract<E, { type: TType }>, Ctx>,
69
+ ): () => void;
70
+
71
+ emit<TEvent extends E>(
72
+ event: TEvent,
73
+ signal?: AbortSignal,
74
+ ): Promise<ResultOf<TEvent> | undefined>;
75
+
76
+ addCleanup(cleanup: () => void | Promise<void>): () => void;
77
+
78
+ clear(): Promise<void>;
79
+ dispose(): Promise<void>;
80
+ }
81
+ ```
82
+
83
+ Important split:
84
+
85
+ - `observe()` sees all events, read-only, return ignored.
86
+ - `on(type, handler)` participates in that event’s semantics.
87
+ - `emit(event)` is the only thing `AgentHarness` calls.
88
+ - `clear()` removes observers/handlers and runs cleanups.
89
+
90
+ ## Default implementation internals
91
+
92
+ ```ts
93
+ class DefaultAgentHarnessHooks<E extends HookEvent<string, unknown>, Ctx>
94
+ implements AgentHarnessHooks<E, Ctx> {
95
+ context: Ctx;
96
+
97
+ private observers = new Set<HookObserver<E, Ctx>>();
98
+ private handlers = new Map<string, Set<HookHandler<any, Ctx>>>();
99
+ private cleanups = new Set<() => void | Promise<void>>();
100
+
101
+ constructor(ctx: Ctx) {
102
+ this.context = ctx;
103
+ }
104
+
105
+ setContext(ctx: Ctx): void {
106
+ this.context = ctx;
107
+ }
108
+
109
+ observe(handler: HookObserver<E, Ctx>): () => void {
110
+ this.observers.add(handler);
111
+ return () => this.observers.delete(handler);
112
+ }
113
+
114
+ on(type, handler): () => void {
115
+ let handlers = this.handlers.get(type);
116
+ if (!handlers) {
117
+ handlers = new Set();
118
+ this.handlers.set(type, handlers);
119
+ }
120
+ handlers.add(handler);
121
+ return () => handlers.delete(handler);
122
+ }
123
+
124
+ async emit(event, signal?) {
125
+ for (const observer of this.observers) {
126
+ await observer(event, this.context, signal);
127
+ }
128
+
129
+ switch (event.type) {
130
+ case "context":
131
+ return this.emitContext(event, signal);
132
+ case "before_provider_request":
133
+ return this.emitBeforeProviderRequest(event, signal);
134
+ case "before_provider_payload":
135
+ return this.emitBeforeProviderPayload(event, signal);
136
+ case "before_agent_start":
137
+ return this.emitBeforeAgentStart(event, signal);
138
+ case "tool_call":
139
+ return this.emitToolCall(event, signal);
140
+ case "tool_result":
141
+ return this.emitToolResult(event, signal);
142
+ case "session_before_compact":
143
+ case "session_before_tree":
144
+ return this.emitFirstCancelOrLast(event, signal);
145
+ default:
146
+ await this.emitObservationHandlers(event, signal);
147
+ return undefined;
148
+ }
149
+ }
150
+ }
151
+ ```
152
+
153
+ Internal casts are acceptable inside the implementation because `Map<string, ...>` loses specificity. Public API remains typed.
154
+
155
+ ## Mutation semantics
156
+
157
+ ### Observation
158
+
159
+ ```ts
160
+ await hooks.emit({ type: "message_end", message }, signal);
161
+ ```
162
+
163
+ Observers run. `message_end` handlers run. Return ignored unless that event later gets a result type.
164
+
165
+ ### Context transform
166
+
167
+ Handlers run in order. Each sees current messages.
168
+
169
+ ```ts
170
+ let current = event;
171
+
172
+ for (const handler of handlers("context")) {
173
+ const result = await handler(current, ctx, signal);
174
+ if (result?.messages) {
175
+ current = { ...current, messages: result.messages };
176
+ }
177
+ }
178
+
179
+ return current.messages === event.messages ? undefined : { messages: current.messages };
180
+ ```
181
+
182
+ ### Provider request / payload
183
+
184
+ Sequential transform. Each handler sees previous output.
185
+
186
+ ```ts
187
+ let current = event;
188
+
189
+ for (const handler of handlers("before_provider_payload")) {
190
+ const result = await handler(current, ctx, signal);
191
+ if (result !== undefined) {
192
+ current = { ...current, payload: result.payload };
193
+ }
194
+ }
195
+
196
+ return changed ? { payload: current.payload } : undefined;
197
+ ```
198
+
199
+ ### Before agent start
200
+
201
+ Collect injected messages, chain system prompt.
202
+
203
+ ```ts
204
+ let systemPrompt = event.systemPrompt;
205
+ const messages = [];
206
+
207
+ for (const handler of handlers("before_agent_start")) {
208
+ const result = await handler({ ...event, systemPrompt }, ctx, signal);
209
+ if (result?.messages) messages.push(...result.messages);
210
+ if (result?.systemPrompt !== undefined) systemPrompt = result.systemPrompt;
211
+ }
212
+
213
+ return messages.length || systemPrompt !== event.systemPrompt
214
+ ? { messages, systemPrompt }
215
+ : undefined;
216
+ ```
217
+
218
+ ### Tool call
219
+
220
+ Sequential, early exit on block.
221
+
222
+ ```ts
223
+ for (const handler of handlers("tool_call")) {
224
+ const result = await handler(event, ctx, signal);
225
+ if (result?.block) return result;
226
+ }
227
+ ```
228
+
229
+ ### Tool result
230
+
231
+ Sequential patch accumulation. Each handler sees current patched result.
232
+
233
+ ```ts
234
+ let current = event;
235
+ let modified = false;
236
+
237
+ for (const handler of handlers("tool_result")) {
238
+ const result = await handler(current, ctx, signal);
239
+ if (!result) continue;
240
+
241
+ current = {
242
+ ...current,
243
+ content: result.content ?? current.content,
244
+ details: result.details ?? current.details,
245
+ isError: result.isError ?? current.isError,
246
+ };
247
+
248
+ modified = true;
249
+ }
250
+
251
+ return modified
252
+ ? { content: current.content, details: current.details, isError: current.isError }
253
+ : undefined;
254
+ ```
255
+
256
+ ### Session-before events
257
+
258
+ Sequential, early exit on cancel.
259
+
260
+ ```ts
261
+ let last;
262
+
263
+ for (const handler of handlers(event.type)) {
264
+ const result = await handler(event, ctx, signal);
265
+ if (!result) continue;
266
+ last = result;
267
+ if (result.cancel) return result;
268
+ }
269
+
270
+ return last;
271
+ ```
272
+
273
+ ## Harness usage
274
+
275
+ Harness only does this:
276
+
277
+ ```ts
278
+ await this.hooks.emit(event, signal);
279
+ ```
280
+
281
+ or:
282
+
283
+ ```ts
284
+ const result = await this.hooks.emit({ type: "context", messages }, signal);
285
+ return result?.messages ?? messages;
286
+ ```
287
+
288
+ Harness does not store handlers, chain listeners, or know extension policy.
289
+
290
+ ## Context
291
+
292
+ Context is a normal object, not rebuilt per emit.
293
+
294
+ ```ts
295
+ const hooks = new CodingAgentHooks({
296
+ harness: harnessFacade,
297
+ session: sessionFacade,
298
+ ui: noUiFacade,
299
+ });
300
+ ```
301
+
302
+ Later:
303
+
304
+ ```ts
305
+ hooks.setContext({
306
+ ...hooks.context,
307
+ ui: tuiFacade,
308
+ });
309
+ ```
310
+
311
+ For dynamic state, prefer stable facades/methods over getter maze:
312
+
313
+ ```ts
314
+ interface CodingAgentHookContext {
315
+ harness: HarnessFacade;
316
+ session: SessionFacade;
317
+ ui: UiFacade;
318
+ models: ModelFacade;
319
+ }
320
+ ```
321
+
322
+ Per-run `signal` is passed as the third handler arg.
323
+
324
+ ## Extension loading later
325
+
326
+ Extension loading can live next to harness and construct hooks:
327
+
328
+ ```ts
329
+ const hooks = await loadExtensions({
330
+ paths,
331
+ context,
332
+ hooks: new CodingAgentHooks(context),
333
+ });
334
+ const harness = new AgentHarness({ ..., hooks });
335
+ ```
336
+
337
+ The loader registers into hooks:
338
+
339
+ ```ts
340
+ hooks.on("context", handler);
341
+ hooks.on("tool_call", handler);
342
+ hooks.addCleanup(cleanup);
343
+ ```
344
+
345
+ For reload:
346
+
347
+ ```ts
348
+ await hooks.clear();
349
+ const nextHooks = await loadExtensions(...);
350
+ harness.setHooks(nextHooks); // idle-only if supported
351
+ ```
352
+
353
+ ## Poking holes
354
+
355
+ ### 1. Error policy must be explicit
356
+
357
+ Existing coding-agent catches extension errors, reports them, and continues. New hooks need the same policy, likely:
358
+
359
+ ```ts
360
+ errorMode: "continue" | "throw"
361
+ onError(error)
362
+ ```
363
+
364
+ For coding-agent, default should be `"continue"`.
365
+
366
+ ### 2. Source metadata matters
367
+
368
+ Existing runner knows which extension produced an error/resource/tool. Plain `on()` loses that unless we add registration metadata or scopes.
369
+
370
+ Probably needed:
371
+
372
+ ```ts
373
+ const scope = hooks.createScope({ sourceInfo });
374
+ scope.on("context", handler);
375
+ scope.addCleanup(...);
376
+ ```
377
+
378
+ Or `on(type, handler, { sourceInfo })`.
379
+
380
+ ### 3. Some extension capabilities are registries, not hooks
381
+
382
+ These are not covered by `emit()` and should stay as registries on `CodingAgentHooks` or an extension host:
383
+
384
+ - tools
385
+ - commands
386
+ - shortcuts
387
+ - flags
388
+ - message renderers
389
+ - provider registrations
390
+ - OAuth providers
391
+ - custom model providers
392
+
393
+ That is fine. They do not belong in `AgentHarness`.
394
+
395
+ ### 4. Existing coding-agent events can be represented
396
+
397
+ No blocker for:
398
+
399
+ - `context`
400
+ - `before_provider_request`
401
+ - `after_provider_response`
402
+ - `before_agent_start`
403
+ - `message_end`
404
+ - `tool_call`
405
+ - `tool_result`
406
+ - `input`
407
+ - `user_bash`
408
+ - `resources_discover`
409
+ - `session_before_*`
410
+ - `session_*`
411
+ - model/thinking selection events
412
+ - agent/turn/message/tool lifecycle events
413
+
414
+ They become additional event types handled by `CodingAgentHooks`.
415
+
416
+ ### 5. Need to preserve exact old semantics
417
+
418
+ When porting coding-agent, special cases must be copied:
419
+
420
+ - `input`: transform chain, `handled` short-circuits.
421
+ - `user_bash`: first meaningful result wins.
422
+ - `message_end`: replacement must keep same role.
423
+ - `before_agent_start`: `ctx.getSystemPrompt()` must reflect current chained prompt.
424
+ - `resources_discover`: aggregate paths and keep extension source.
425
+ - `tool_call`: argument mutation remains visible to later handlers.
426
+ - `tool_result`: later handlers see prior patches.
427
+
428
+ The design allows all of that, but the default/coding hooks implementation must encode it.
429
+
430
+ ### 6. `emit()` switch can miss custom mutation events
431
+
432
+ If a subclass adds a result-producing event but forgets to override `emit()`, it will behave observationally. Tests should catch this. Could add a protected strategy registry later if this becomes error-prone, but not initially.
433
+
434
+ ### 7. Observer semantics are intentionally limited
435
+
436
+ Observers see the original emitted event once. They do not see every intermediate mutation. If something needs final transformed state, emit a separate final event or use an event-specific handler.
437
+
438
+ ## Verdict
439
+
440
+ This design can implement a new coding-agent. It is simpler than the current runner, keeps harness clean, and preserves the important extension capabilities as long as `CodingAgentHooks` adds source-aware scopes, registries, cleanup, and the exact old event semantics.
441
+
442
+ --- Comments ---
443
+
444
+ Thread hn2xk0tzhj on "addCleanup(cleanup"
445
+ [tmluyaub9v] Owner (2026-05-14T12:55:45.500Z): cleanup should be passed along optionally to on/observe