world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,747 @@
1
+ import type { AssistantMessage, ImageContent, Model, Models, TextContent, Usage } from "@earendil-works/pi-ai";
2
+ import type { AgentMessage, ThinkingLevel } from "../../types.ts";
3
+ import {
4
+ convertToLlm,
5
+ createBranchSummaryMessage,
6
+ createCompactionSummaryMessage,
7
+ createCustomMessage,
8
+ } from "../messages.ts";
9
+ import { buildSessionContext } from "../session/session.ts";
10
+ import { type CompactionEntry, CompactionError, err, ok, type Result, type SessionTreeEntry } from "../types.ts";
11
+ import {
12
+ computeFileLists,
13
+ createFileOps,
14
+ extractFileOpsFromMessage,
15
+ type FileOperations,
16
+ formatFileOperations,
17
+ serializeConversation,
18
+ } from "./utils.ts";
19
+
20
+ /** File-operation details stored on generated compaction entries. */
21
+ export interface CompactionDetails {
22
+ /** Files read in the compacted history. */
23
+ readFiles: string[];
24
+ /** Files modified in the compacted history. */
25
+ modifiedFiles: string[];
26
+ }
27
+ function safeJsonStringify(value: unknown): string {
28
+ try {
29
+ return JSON.stringify(value) ?? "undefined";
30
+ } catch {
31
+ return "[unserializable]";
32
+ }
33
+ }
34
+
35
+ function extractFileOperations(
36
+ messages: AgentMessage[],
37
+ entries: SessionTreeEntry[],
38
+ prevCompactionIndex: number,
39
+ ): FileOperations {
40
+ const fileOps = createFileOps();
41
+ if (prevCompactionIndex >= 0) {
42
+ const prevCompaction = entries[prevCompactionIndex] as CompactionEntry;
43
+ if (!prevCompaction.fromHook && prevCompaction.details) {
44
+ const details = prevCompaction.details as CompactionDetails;
45
+ if (Array.isArray(details.readFiles)) {
46
+ for (const f of details.readFiles) fileOps.read.add(f);
47
+ }
48
+ if (Array.isArray(details.modifiedFiles)) {
49
+ for (const f of details.modifiedFiles) fileOps.edited.add(f);
50
+ }
51
+ }
52
+ }
53
+ for (const msg of messages) {
54
+ extractFileOpsFromMessage(msg, fileOps);
55
+ }
56
+
57
+ return fileOps;
58
+ }
59
+ function getMessageFromEntry(entry: SessionTreeEntry): AgentMessage | undefined {
60
+ if (entry.type === "message") {
61
+ return entry.message as AgentMessage;
62
+ }
63
+ if (entry.type === "custom_message") {
64
+ return createCustomMessage(
65
+ entry.customType,
66
+ entry.content as string | (TextContent | ImageContent)[],
67
+ entry.display,
68
+ entry.details,
69
+ entry.timestamp,
70
+ );
71
+ }
72
+ if (entry.type === "branch_summary") {
73
+ return createBranchSummaryMessage(entry.summary, entry.fromId, entry.timestamp);
74
+ }
75
+ if (entry.type === "compaction") {
76
+ return createCompactionSummaryMessage(entry.summary, entry.tokensBefore, entry.timestamp);
77
+ }
78
+ return undefined;
79
+ }
80
+
81
+ function getMessageFromEntryForCompaction(entry: SessionTreeEntry): AgentMessage | undefined {
82
+ if (entry.type === "compaction") {
83
+ return undefined;
84
+ }
85
+ return getMessageFromEntry(entry);
86
+ }
87
+
88
+ /** Generated compaction data ready to be persisted as a compaction entry. */
89
+ export interface CompactionResult<T = unknown> {
90
+ /** Summary text that replaces compacted history in future context. */
91
+ summary: string;
92
+ /** Entry id where retained history starts. */
93
+ firstKeptEntryId: string;
94
+ /** Estimated context tokens before compaction. */
95
+ tokensBefore: number;
96
+ /** Optional implementation-specific details stored with the compaction entry. */
97
+ details?: T;
98
+ }
99
+
100
+ /** Compaction thresholds and retention settings. */
101
+ export interface CompactionSettings {
102
+ /** Enable automatic compaction decisions. */
103
+ enabled: boolean;
104
+ /** Tokens reserved for summary prompt and output. */
105
+ reserveTokens: number;
106
+ /** Approximate recent-context tokens to keep after compaction. */
107
+ keepRecentTokens: number;
108
+ }
109
+
110
+ /** Default compaction settings used by the harness. */
111
+ export const DEFAULT_COMPACTION_SETTINGS: CompactionSettings = {
112
+ enabled: true,
113
+ reserveTokens: 16384,
114
+ keepRecentTokens: 20000,
115
+ };
116
+
117
+ /** Calculate total context tokens from provider usage. */
118
+ export function calculateContextTokens(usage: Usage): number {
119
+ return usage.totalTokens || usage.input + usage.output + usage.cacheRead + usage.cacheWrite;
120
+ }
121
+ function getAssistantUsage(msg: AgentMessage): Usage | undefined {
122
+ if (msg.role === "assistant" && "usage" in msg) {
123
+ const assistantMsg = msg as AssistantMessage;
124
+ if (
125
+ assistantMsg.stopReason !== "aborted" &&
126
+ assistantMsg.stopReason !== "error" &&
127
+ assistantMsg.usage &&
128
+ calculateContextTokens(assistantMsg.usage) > 0
129
+ ) {
130
+ return assistantMsg.usage;
131
+ }
132
+ }
133
+ return undefined;
134
+ }
135
+
136
+ /** Return usage from the last valid assistant message in session entries. */
137
+ export function getLastAssistantUsage(entries: SessionTreeEntry[]): Usage | undefined {
138
+ for (let i = entries.length - 1; i >= 0; i--) {
139
+ const entry = entries[i];
140
+ if (entry.type === "message") {
141
+ const usage = getAssistantUsage(entry.message as AgentMessage);
142
+ if (usage) return usage;
143
+ }
144
+ }
145
+ return undefined;
146
+ }
147
+
148
+ /** Estimated context-token usage for a message list. */
149
+ export interface ContextUsageEstimate {
150
+ /** Estimated total context tokens. */
151
+ tokens: number;
152
+ /** Tokens reported by the most recent assistant usage block. */
153
+ usageTokens: number;
154
+ /** Estimated tokens after the most recent assistant usage block. */
155
+ trailingTokens: number;
156
+ /** Index of the message that provided usage, or null when none exists. */
157
+ lastUsageIndex: number | null;
158
+ }
159
+
160
+ function getLastAssistantUsageInfo(messages: AgentMessage[]): { usage: Usage; index: number } | undefined {
161
+ for (let i = messages.length - 1; i >= 0; i--) {
162
+ const usage = getAssistantUsage(messages[i]);
163
+ if (usage) return { usage, index: i };
164
+ }
165
+ return undefined;
166
+ }
167
+
168
+ /** Estimate context tokens for messages using provider usage when available. */
169
+ export function estimateContextTokens(messages: AgentMessage[]): ContextUsageEstimate {
170
+ const usageInfo = getLastAssistantUsageInfo(messages);
171
+
172
+ if (!usageInfo) {
173
+ let estimated = 0;
174
+ for (const message of messages) {
175
+ estimated += estimateTokens(message);
176
+ }
177
+ return {
178
+ tokens: estimated,
179
+ usageTokens: 0,
180
+ trailingTokens: estimated,
181
+ lastUsageIndex: null,
182
+ };
183
+ }
184
+
185
+ const usageTokens = calculateContextTokens(usageInfo.usage);
186
+ let trailingTokens = 0;
187
+ for (let i = usageInfo.index + 1; i < messages.length; i++) {
188
+ trailingTokens += estimateTokens(messages[i]);
189
+ }
190
+
191
+ return {
192
+ tokens: usageTokens + trailingTokens,
193
+ usageTokens,
194
+ trailingTokens,
195
+ lastUsageIndex: usageInfo.index,
196
+ };
197
+ }
198
+
199
+ /** Return whether context usage exceeds the configured compaction threshold. */
200
+ export function shouldCompact(contextTokens: number, contextWindow: number, settings: CompactionSettings): boolean {
201
+ if (!settings.enabled) return false;
202
+ return contextTokens > contextWindow - settings.reserveTokens;
203
+ }
204
+
205
+ const ESTIMATED_IMAGE_CHARS = 4800;
206
+
207
+ function estimateTextAndImageContentChars(content: string | Array<{ type: string; text?: string }>): number {
208
+ if (typeof content === "string") {
209
+ return content.length;
210
+ }
211
+
212
+ let chars = 0;
213
+ for (const block of content) {
214
+ if (block.type === "text" && block.text) {
215
+ chars += block.text.length;
216
+ } else if (block.type === "image") {
217
+ chars += ESTIMATED_IMAGE_CHARS;
218
+ }
219
+ }
220
+ return chars;
221
+ }
222
+
223
+ /** Estimate token count for one message using a conservative character heuristic. */
224
+ export function estimateTokens(message: AgentMessage): number {
225
+ let chars = 0;
226
+
227
+ switch (message.role) {
228
+ case "user": {
229
+ chars = estimateTextAndImageContentChars(
230
+ (message as { content: string | Array<{ type: string; text?: string }> }).content,
231
+ );
232
+ return Math.ceil(chars / 4);
233
+ }
234
+ case "assistant": {
235
+ const assistant = message as AssistantMessage;
236
+ for (const block of assistant.content) {
237
+ if (block.type === "text") {
238
+ chars += block.text.length;
239
+ } else if (block.type === "thinking") {
240
+ chars += block.thinking.length;
241
+ } else if (block.type === "toolCall") {
242
+ chars += block.name.length + safeJsonStringify(block.arguments).length;
243
+ }
244
+ }
245
+ return Math.ceil(chars / 4);
246
+ }
247
+ case "custom":
248
+ case "toolResult": {
249
+ chars = estimateTextAndImageContentChars(message.content);
250
+ return Math.ceil(chars / 4);
251
+ }
252
+ case "bashExecution": {
253
+ chars = message.command.length + message.output.length;
254
+ return Math.ceil(chars / 4);
255
+ }
256
+ case "branchSummary":
257
+ case "compactionSummary": {
258
+ chars = message.summary.length;
259
+ return Math.ceil(chars / 4);
260
+ }
261
+ }
262
+
263
+ return 0;
264
+ }
265
+ function findValidCutPoints(entries: SessionTreeEntry[], startIndex: number, endIndex: number): number[] {
266
+ const cutPoints: number[] = [];
267
+ for (let i = startIndex; i < endIndex; i++) {
268
+ const entry = entries[i];
269
+ switch (entry.type) {
270
+ case "message": {
271
+ const role = entry.message.role;
272
+ switch (role) {
273
+ case "bashExecution":
274
+ case "custom":
275
+ case "branchSummary":
276
+ case "compactionSummary":
277
+ case "user":
278
+ case "assistant":
279
+ cutPoints.push(i);
280
+ break;
281
+ case "toolResult":
282
+ break;
283
+ }
284
+ break;
285
+ }
286
+ case "thinking_level_change":
287
+ case "model_change":
288
+ case "active_tools_change":
289
+ case "compaction":
290
+ case "branch_summary":
291
+ case "custom":
292
+ case "custom_message":
293
+ case "label":
294
+ case "session_info":
295
+ case "leaf":
296
+ break;
297
+ }
298
+ if (entry.type === "branch_summary" || entry.type === "custom_message") {
299
+ cutPoints.push(i);
300
+ }
301
+ }
302
+ return cutPoints;
303
+ }
304
+
305
+ /** Find the user-visible message that starts the turn containing an entry. */
306
+ export function findTurnStartIndex(entries: SessionTreeEntry[], entryIndex: number, startIndex: number): number {
307
+ for (let i = entryIndex; i >= startIndex; i--) {
308
+ const entry = entries[i];
309
+ if (entry.type === "branch_summary" || entry.type === "custom_message") {
310
+ return i;
311
+ }
312
+ if (entry.type === "message") {
313
+ const role = entry.message.role;
314
+ if (role === "user" || role === "bashExecution") {
315
+ return i;
316
+ }
317
+ }
318
+ }
319
+ return -1;
320
+ }
321
+
322
+ /** Cut point selected for compaction. */
323
+ export interface CutPointResult {
324
+ /** Index of the first entry retained after compaction. */
325
+ firstKeptEntryIndex: number;
326
+ /** Index of the turn-start entry when the cut splits a turn, otherwise -1. */
327
+ turnStartIndex: number;
328
+ /** Whether the selected cut point splits an in-progress turn. */
329
+ isSplitTurn: boolean;
330
+ }
331
+
332
+ /** Find the compaction cut point that keeps approximately the requested recent-token budget. */
333
+ export function findCutPoint(
334
+ entries: SessionTreeEntry[],
335
+ startIndex: number,
336
+ endIndex: number,
337
+ keepRecentTokens: number,
338
+ ): CutPointResult {
339
+ const cutPoints = findValidCutPoints(entries, startIndex, endIndex);
340
+
341
+ if (cutPoints.length === 0) {
342
+ return { firstKeptEntryIndex: startIndex, turnStartIndex: -1, isSplitTurn: false };
343
+ }
344
+ let accumulatedTokens = 0;
345
+ let cutIndex = cutPoints[0];
346
+
347
+ for (let i = endIndex - 1; i >= startIndex; i--) {
348
+ const entry = entries[i];
349
+ if (entry.type !== "message") continue;
350
+ const messageTokens = estimateTokens(entry.message as AgentMessage);
351
+ accumulatedTokens += messageTokens;
352
+ if (accumulatedTokens >= keepRecentTokens) {
353
+ for (let c = 0; c < cutPoints.length; c++) {
354
+ if (cutPoints[c] >= i) {
355
+ cutIndex = cutPoints[c];
356
+ break;
357
+ }
358
+ }
359
+ break;
360
+ }
361
+ }
362
+ while (cutIndex > startIndex) {
363
+ const prevEntry = entries[cutIndex - 1];
364
+ if (prevEntry.type === "compaction") {
365
+ break;
366
+ }
367
+ if (prevEntry.type === "message") {
368
+ break;
369
+ }
370
+ cutIndex--;
371
+ }
372
+ const cutEntry = entries[cutIndex];
373
+ const isUserMessage = cutEntry.type === "message" && cutEntry.message.role === "user";
374
+ const turnStartIndex = isUserMessage ? -1 : findTurnStartIndex(entries, cutIndex, startIndex);
375
+
376
+ return {
377
+ firstKeptEntryIndex: cutIndex,
378
+ turnStartIndex,
379
+ isSplitTurn: !isUserMessage && turnStartIndex !== -1,
380
+ };
381
+ }
382
+
383
+ export const SUMMARIZATION_SYSTEM_PROMPT = `You are a context summarization assistant. Your task is to read a conversation between a user and an AI assistant, then produce a structured summary following the exact format specified.
384
+
385
+ Do NOT continue the conversation. Do NOT respond to any questions in the conversation. ONLY output the structured summary.`;
386
+
387
+ const SUMMARIZATION_PROMPT = `The messages above are a conversation to summarize. Create a structured context checkpoint summary that another LLM will use to continue the work.
388
+
389
+ Use this EXACT format:
390
+
391
+ ## Goal
392
+ [What is the user trying to accomplish? Can be multiple items if the session covers different tasks.]
393
+
394
+ ## Constraints & Preferences
395
+ - [Any constraints, preferences, or requirements mentioned by user]
396
+ - [Or "(none)" if none were mentioned]
397
+
398
+ ## Progress
399
+ ### Done
400
+ - [x] [Completed tasks/changes]
401
+
402
+ ### In Progress
403
+ - [ ] [Current work]
404
+
405
+ ### Blocked
406
+ - [Issues preventing progress, if any]
407
+
408
+ ## Key Decisions
409
+ - **[Decision]**: [Brief rationale]
410
+
411
+ ## Next Steps
412
+ 1. [Ordered list of what should happen next]
413
+
414
+ ## Critical Context
415
+ - [Any data, examples, or references needed to continue]
416
+ - [Or "(none)" if not applicable]
417
+
418
+ Keep each section concise. Preserve exact file paths, function names, and error messages.`;
419
+
420
+ const UPDATE_SUMMARIZATION_PROMPT = `The messages above are NEW conversation messages to incorporate into the existing summary provided in <previous-summary> tags.
421
+
422
+ Update the existing structured summary with new information. RULES:
423
+ - PRESERVE all existing information from the previous summary
424
+ - ADD new progress, decisions, and context from the new messages
425
+ - UPDATE the Progress section: move items from "In Progress" to "Done" when completed
426
+ - UPDATE "Next Steps" based on what was accomplished
427
+ - PRESERVE exact file paths, function names, and error messages
428
+ - If something is no longer relevant, you may remove it
429
+
430
+ Use this EXACT format:
431
+
432
+ ## Goal
433
+ [Preserve existing goals, add new ones if the task expanded]
434
+
435
+ ## Constraints & Preferences
436
+ - [Preserve existing, add new ones discovered]
437
+
438
+ ## Progress
439
+ ### Done
440
+ - [x] [Include previously done items AND newly completed items]
441
+
442
+ ### In Progress
443
+ - [ ] [Current work - update based on progress]
444
+
445
+ ### Blocked
446
+ - [Current blockers - remove if resolved]
447
+
448
+ ## Key Decisions
449
+ - **[Decision]**: [Brief rationale] (preserve all previous, add new)
450
+
451
+ ## Next Steps
452
+ 1. [Update based on current state]
453
+
454
+ ## Critical Context
455
+ - [Preserve important context, add new if needed]
456
+
457
+ Keep each section concise. Preserve exact file paths, function names, and error messages.`;
458
+
459
+ /** Generate or update a conversation summary for compaction. */
460
+ export async function generateSummary(
461
+ currentMessages: AgentMessage[],
462
+ models: Models,
463
+ model: Model<any>,
464
+ reserveTokens: number,
465
+ signal?: AbortSignal,
466
+ customInstructions?: string,
467
+ previousSummary?: string,
468
+ thinkingLevel?: ThinkingLevel,
469
+ ): Promise<Result<string, CompactionError>> {
470
+ const maxTokens = Math.min(
471
+ Math.floor(0.8 * reserveTokens),
472
+ model.maxTokens > 0 ? model.maxTokens : Number.POSITIVE_INFINITY,
473
+ );
474
+ let basePrompt = previousSummary ? UPDATE_SUMMARIZATION_PROMPT : SUMMARIZATION_PROMPT;
475
+ if (customInstructions) {
476
+ basePrompt = `${basePrompt}\n\nAdditional focus: ${customInstructions}`;
477
+ }
478
+ const llmMessages = convertToLlm(currentMessages);
479
+ const conversationText = serializeConversation(llmMessages);
480
+ let promptText = `<conversation>\n${conversationText}\n</conversation>\n\n`;
481
+ if (previousSummary) {
482
+ promptText += `<previous-summary>\n${previousSummary}\n</previous-summary>\n\n`;
483
+ }
484
+ promptText += basePrompt;
485
+
486
+ const summarizationMessages = [
487
+ {
488
+ role: "user" as const,
489
+ content: [{ type: "text" as const, text: promptText }],
490
+ timestamp: Date.now(),
491
+ },
492
+ ];
493
+
494
+ const completionOptions =
495
+ model.reasoning && thinkingLevel && thinkingLevel !== "off"
496
+ ? { maxTokens, signal, reasoning: thinkingLevel }
497
+ : { maxTokens, signal };
498
+
499
+ const response = await models.completeSimple(
500
+ model,
501
+ { systemPrompt: SUMMARIZATION_SYSTEM_PROMPT, messages: summarizationMessages },
502
+ completionOptions,
503
+ );
504
+ if (response.stopReason === "aborted") {
505
+ return err(new CompactionError("aborted", response.errorMessage || "Summarization aborted"));
506
+ }
507
+ if (response.stopReason === "error") {
508
+ return err(
509
+ new CompactionError(
510
+ "summarization_failed",
511
+ `Summarization failed: ${response.errorMessage || "Unknown error"}`,
512
+ ),
513
+ );
514
+ }
515
+
516
+ const textContent = response.content
517
+ .filter((c): c is { type: "text"; text: string } => c.type === "text")
518
+ .map((c) => c.text)
519
+ .join("\n");
520
+
521
+ return ok(textContent);
522
+ }
523
+
524
+ /** Prepared inputs for a compaction run. */
525
+ export interface CompactionPreparation {
526
+ /** Entry id where retained history starts. */
527
+ firstKeptEntryId: string;
528
+ /** Messages summarized into the history summary. */
529
+ messagesToSummarize: AgentMessage[];
530
+ /** Prefix messages summarized separately when compaction splits a turn. */
531
+ turnPrefixMessages: AgentMessage[];
532
+ /** Whether compaction splits a turn. */
533
+ isSplitTurn: boolean;
534
+ /** Estimated context tokens before compaction. */
535
+ tokensBefore: number;
536
+ /** Previous compaction summary used for iterative updates. */
537
+ previousSummary?: string;
538
+ /** File operations extracted from summarized history. */
539
+ fileOps: FileOperations;
540
+ /** Settings used to prepare compaction. */
541
+ settings: CompactionSettings;
542
+ }
543
+
544
+ /** Prepare session entries for compaction, or return undefined when compaction is not applicable. */
545
+ export function prepareCompaction(
546
+ pathEntries: SessionTreeEntry[],
547
+ settings: CompactionSettings,
548
+ ): Result<CompactionPreparation | undefined, CompactionError> {
549
+ if (pathEntries.length === 0 || pathEntries[pathEntries.length - 1].type === "compaction") {
550
+ return ok(undefined);
551
+ }
552
+
553
+ let prevCompactionIndex = -1;
554
+ for (let i = pathEntries.length - 1; i >= 0; i--) {
555
+ if (pathEntries[i].type === "compaction") {
556
+ prevCompactionIndex = i;
557
+ break;
558
+ }
559
+ }
560
+
561
+ let previousSummary: string | undefined;
562
+ let boundaryStart = 0;
563
+ if (prevCompactionIndex >= 0) {
564
+ const prevCompaction = pathEntries[prevCompactionIndex] as CompactionEntry;
565
+ previousSummary = prevCompaction.summary;
566
+ const firstKeptEntryIndex = pathEntries.findIndex((entry) => entry.id === prevCompaction.firstKeptEntryId);
567
+ boundaryStart = firstKeptEntryIndex >= 0 ? firstKeptEntryIndex : prevCompactionIndex + 1;
568
+ }
569
+ const boundaryEnd = pathEntries.length;
570
+
571
+ const tokensBefore = estimateContextTokens(buildSessionContext(pathEntries).messages).tokens;
572
+
573
+ const cutPoint = findCutPoint(pathEntries, boundaryStart, boundaryEnd, settings.keepRecentTokens);
574
+ const firstKeptEntry = pathEntries[cutPoint.firstKeptEntryIndex];
575
+ if (!firstKeptEntry?.id) {
576
+ return err(new CompactionError("invalid_session", "First kept entry has no UUID - session may need migration"));
577
+ }
578
+ const firstKeptEntryId = firstKeptEntry.id;
579
+
580
+ const historyEnd = cutPoint.isSplitTurn ? cutPoint.turnStartIndex : cutPoint.firstKeptEntryIndex;
581
+ const messagesToSummarize: AgentMessage[] = [];
582
+ for (let i = boundaryStart; i < historyEnd; i++) {
583
+ const msg = getMessageFromEntryForCompaction(pathEntries[i]);
584
+ if (msg) messagesToSummarize.push(msg);
585
+ }
586
+ const turnPrefixMessages: AgentMessage[] = [];
587
+ if (cutPoint.isSplitTurn) {
588
+ for (let i = cutPoint.turnStartIndex; i < cutPoint.firstKeptEntryIndex; i++) {
589
+ const msg = getMessageFromEntryForCompaction(pathEntries[i]);
590
+ if (msg) turnPrefixMessages.push(msg);
591
+ }
592
+ }
593
+ const fileOps = extractFileOperations(messagesToSummarize, pathEntries, prevCompactionIndex);
594
+ if (cutPoint.isSplitTurn) {
595
+ for (const msg of turnPrefixMessages) {
596
+ extractFileOpsFromMessage(msg, fileOps);
597
+ }
598
+ }
599
+
600
+ return ok({
601
+ firstKeptEntryId,
602
+ messagesToSummarize,
603
+ turnPrefixMessages,
604
+ isSplitTurn: cutPoint.isSplitTurn,
605
+ tokensBefore,
606
+ previousSummary,
607
+ fileOps,
608
+ settings,
609
+ });
610
+ }
611
+
612
+ const TURN_PREFIX_SUMMARIZATION_PROMPT = `This is the PREFIX of a turn that was too large to keep. The SUFFIX (recent work) is retained.
613
+
614
+ Summarize the prefix to provide context for the retained suffix:
615
+
616
+ ## Original Request
617
+ [What did the user ask for in this turn?]
618
+
619
+ ## Early Progress
620
+ - [Key decisions and work done in the prefix]
621
+
622
+ ## Context for Suffix
623
+ - [Information needed to understand the retained recent work]
624
+
625
+ Be concise. Focus on what's needed to understand the kept suffix.`;
626
+
627
+ export { serializeConversation } from "./utils.ts";
628
+
629
+ /** Generate compaction summary data from prepared session history. */
630
+ export async function compact(
631
+ preparation: CompactionPreparation,
632
+ models: Models,
633
+ model: Model<any>,
634
+ customInstructions?: string,
635
+ signal?: AbortSignal,
636
+ thinkingLevel?: ThinkingLevel,
637
+ ): Promise<Result<CompactionResult, CompactionError>> {
638
+ const {
639
+ firstKeptEntryId,
640
+ messagesToSummarize,
641
+ turnPrefixMessages,
642
+ isSplitTurn,
643
+ tokensBefore,
644
+ previousSummary,
645
+ fileOps,
646
+ settings,
647
+ } = preparation;
648
+
649
+ if (!firstKeptEntryId) {
650
+ return err(new CompactionError("invalid_session", "First kept entry has no UUID - session may need migration"));
651
+ }
652
+
653
+ let summary: string;
654
+
655
+ if (isSplitTurn && turnPrefixMessages.length > 0) {
656
+ const [historyResult, turnPrefixResult] = await Promise.all([
657
+ messagesToSummarize.length > 0
658
+ ? generateSummary(
659
+ messagesToSummarize,
660
+ models,
661
+ model,
662
+ settings.reserveTokens,
663
+ signal,
664
+ customInstructions,
665
+ previousSummary,
666
+ thinkingLevel,
667
+ )
668
+ : Promise.resolve(ok<string, CompactionError>("No prior history.")),
669
+ generateTurnPrefixSummary(turnPrefixMessages, models, model, settings.reserveTokens, signal, thinkingLevel),
670
+ ]);
671
+ if (!historyResult.ok) return err(historyResult.error);
672
+ if (!turnPrefixResult.ok) return err(turnPrefixResult.error);
673
+ summary = `${historyResult.value}\n\n---\n\n**Turn Context (split turn):**\n\n${turnPrefixResult.value}`;
674
+ } else {
675
+ const summaryResult = await generateSummary(
676
+ messagesToSummarize,
677
+ models,
678
+ model,
679
+ settings.reserveTokens,
680
+ signal,
681
+ customInstructions,
682
+ previousSummary,
683
+ thinkingLevel,
684
+ );
685
+ if (!summaryResult.ok) return err(summaryResult.error);
686
+ summary = summaryResult.value;
687
+ }
688
+
689
+ const { readFiles, modifiedFiles } = computeFileLists(fileOps);
690
+ summary += formatFileOperations(readFiles, modifiedFiles);
691
+
692
+ return ok({
693
+ summary,
694
+ firstKeptEntryId,
695
+ tokensBefore,
696
+ details: { readFiles, modifiedFiles } as CompactionDetails,
697
+ });
698
+ }
699
+ async function generateTurnPrefixSummary(
700
+ messages: AgentMessage[],
701
+ models: Models,
702
+ model: Model<any>,
703
+ reserveTokens: number,
704
+ signal?: AbortSignal,
705
+ thinkingLevel?: ThinkingLevel,
706
+ ): Promise<Result<string, CompactionError>> {
707
+ const maxTokens = Math.min(
708
+ Math.floor(0.5 * reserveTokens),
709
+ model.maxTokens > 0 ? model.maxTokens : Number.POSITIVE_INFINITY,
710
+ );
711
+ const llmMessages = convertToLlm(messages);
712
+ const conversationText = serializeConversation(llmMessages);
713
+ const promptText = `<conversation>\n${conversationText}\n</conversation>\n\n${TURN_PREFIX_SUMMARIZATION_PROMPT}`;
714
+ const summarizationMessages = [
715
+ {
716
+ role: "user" as const,
717
+ content: [{ type: "text" as const, text: promptText }],
718
+ timestamp: Date.now(),
719
+ },
720
+ ];
721
+
722
+ const response = await models.completeSimple(
723
+ model,
724
+ { systemPrompt: SUMMARIZATION_SYSTEM_PROMPT, messages: summarizationMessages },
725
+ model.reasoning && thinkingLevel && thinkingLevel !== "off"
726
+ ? { maxTokens, signal, reasoning: thinkingLevel }
727
+ : { maxTokens, signal },
728
+ );
729
+ if (response.stopReason === "aborted") {
730
+ return err(new CompactionError("aborted", response.errorMessage || "Turn prefix summarization aborted"));
731
+ }
732
+ if (response.stopReason === "error") {
733
+ return err(
734
+ new CompactionError(
735
+ "summarization_failed",
736
+ `Turn prefix summarization failed: ${response.errorMessage || "Unknown error"}`,
737
+ ),
738
+ );
739
+ }
740
+
741
+ return ok(
742
+ response.content
743
+ .filter((c): c is { type: "text"; text: string } => c.type === "text")
744
+ .map((c) => c.text)
745
+ .join("\n"),
746
+ );
747
+ }