@selesai/code 0.13.26 → 0.13.27

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/CHANGELOG.md +74 -0
  2. package/dist/cli/args.js +1 -0
  3. package/dist/cli/session-picker.d.ts +1 -1
  4. package/dist/config.js +1 -1
  5. package/dist/core/agent-session-auto-handoff.test.js +2 -1
  6. package/dist/core/agent-session-length-continuation.test.js +2 -1
  7. package/dist/core/agent-session.d.ts +58 -7
  8. package/dist/core/agent-session.js +334 -167
  9. package/dist/core/bug-report-upload.d.ts +11 -0
  10. package/dist/core/bug-report-upload.js +21 -0
  11. package/dist/core/bug-report.d.ts +175 -0
  12. package/dist/core/bug-report.js +291 -0
  13. package/dist/core/cache-stats.js +12 -1
  14. package/dist/core/cache-warmer.d.ts +101 -0
  15. package/dist/core/cache-warmer.js +341 -0
  16. package/dist/core/compaction/branch-summarization.js +2 -2
  17. package/dist/core/compaction/compaction.d.ts +6 -5
  18. package/dist/core/compaction/compaction.js +49 -42
  19. package/dist/core/crash-log.d.ts +21 -0
  20. package/dist/core/crash-log.js +70 -0
  21. package/dist/core/experimental.d.ts +0 -4
  22. package/dist/core/experimental.js +0 -4
  23. package/dist/core/extensions/index.d.ts +1 -1
  24. package/dist/core/extensions/jiti-loader.d.ts +1 -0
  25. package/dist/core/extensions/jiti-loader.js +3 -0
  26. package/dist/core/extensions/jiti-static-loader.d.ts +1 -0
  27. package/dist/core/extensions/jiti-static-loader.js +3 -0
  28. package/dist/core/extensions/loader.js +39 -49
  29. package/dist/core/extensions/runner.d.ts +10 -7
  30. package/dist/core/extensions/runner.js +85 -67
  31. package/dist/core/extensions/types.d.ts +66 -52
  32. package/dist/core/extensions/virtual-modules.d.ts +2 -0
  33. package/dist/core/extensions/virtual-modules.js +39 -0
  34. package/dist/core/extensions/wrapper.js +1 -20
  35. package/dist/core/index.d.ts +2 -1
  36. package/dist/core/keybindings.d.ts +1 -1
  37. package/dist/core/keybindings.js +1 -1
  38. package/dist/core/messages.js +1 -0
  39. package/dist/core/model-config.d.ts +101 -20
  40. package/dist/core/model-config.js +25 -18
  41. package/dist/core/model-resolver.js +2 -1
  42. package/dist/core/model-runtime.js +5 -3
  43. package/dist/core/provider-composer.d.ts +7 -2
  44. package/dist/core/provider-composer.js +10 -5
  45. package/dist/core/radius.d.ts +3 -0
  46. package/dist/core/radius.js +6 -0
  47. package/dist/core/sdk.js +60 -38
  48. package/dist/core/session-export.d.ts +6 -2
  49. package/dist/core/session-export.js +14 -13
  50. package/dist/core/session-format.test.d.ts +1 -0
  51. package/dist/core/session-format.test.js +96 -0
  52. package/dist/core/session-manager.d.ts +29 -6
  53. package/dist/core/session-manager.js +159 -96
  54. package/dist/core/settings-manager.d.ts +22 -5
  55. package/dist/core/settings-manager.js +56 -23
  56. package/dist/core/slash-commands.js +1 -0
  57. package/dist/core/system-prompt.d.ts +49 -4
  58. package/dist/core/system-prompt.js +155 -102
  59. package/dist/core/system-prompt.test.js +61 -23
  60. package/dist/core/tools/bash.d.ts +2 -1
  61. package/dist/core/tools/bash.js +10 -4
  62. package/dist/core/tools/edit.js +1 -2
  63. package/dist/core/tools/read.js +1 -2
  64. package/dist/core/tools/renderers/bash.js +9 -1
  65. package/dist/core/tools/write.js +1 -2
  66. package/dist/core/usage-totals.d.ts +1 -1
  67. package/dist/core/usage-totals.js +5 -1
  68. package/dist/extensions/pi-web-agent/src/backends/factory.ts +14 -6
  69. package/dist/extensions/pi-web-agent/src/extension.ts +3 -1
  70. package/dist/extensions/pi-web-agent/src/orchestration/answer-synthesizer.ts +7 -3
  71. package/dist/extensions/pi-web-agent/src/orchestration/research-orchestrator.ts +3 -0
  72. package/dist/extensions/pi-web-agent/src/orchestration/research-types.ts +2 -0
  73. package/dist/extensions/pi-web-agent/src/presentation/explore-presentation.ts +4 -2
  74. package/dist/extensions/pi-web-agent/src/search/tokenin.ts +4 -1
  75. package/dist/extensions/pi-web-agent/src/tools/web-explore.ts +2 -1
  76. package/dist/extensions/pi-web-agent/src/types.ts +4 -0
  77. package/dist/index.d.ts +3 -2
  78. package/dist/main.js +13 -9
  79. package/dist/migrations.js +2 -2
  80. package/dist/modes/interactive/bug-report.d.ts +15 -0
  81. package/dist/modes/interactive/bug-report.js +205 -0
  82. package/dist/modes/interactive/chat-viewport.js +1 -1
  83. package/dist/modes/interactive/components/branch-summary-message.js +12 -5
  84. package/dist/modes/interactive/components/compaction-summary-message.js +12 -5
  85. package/dist/modes/interactive/components/custom-editor.d.ts +3 -3
  86. package/dist/modes/interactive/components/extension-editor.d.ts +4 -1
  87. package/dist/modes/interactive/components/extension-editor.js +7 -2
  88. package/dist/modes/interactive/components/extension-input.d.ts +2 -0
  89. package/dist/modes/interactive/components/extension-input.js +6 -0
  90. package/dist/modes/interactive/components/extension-selector.d.ts +1 -0
  91. package/dist/modes/interactive/components/extension-selector.js +4 -0
  92. package/dist/modes/interactive/components/footer.js +4 -1
  93. package/dist/modes/interactive/components/session-selector.d.ts +5 -5
  94. package/dist/modes/interactive/components/session-selector.js +74 -48
  95. package/dist/modes/interactive/components/settings-selector.d.ts +3 -1
  96. package/dist/modes/interactive/components/settings-selector.js +11 -0
  97. package/dist/modes/interactive/components/skill-invocation-message.js +11 -4
  98. package/dist/modes/interactive/components/status-indicator.d.ts +2 -2
  99. package/dist/modes/interactive/components/status-indicator.js +7 -7
  100. package/dist/modes/interactive/components/tool-execution.js +18 -9
  101. package/dist/modes/interactive/components/tree-selector.js +2 -0
  102. package/dist/modes/interactive/interactive-mode.d.ts +15 -2
  103. package/dist/modes/interactive/interactive-mode.js +257 -81
  104. package/dist/modes/interactive/session-share.d.ts +2 -0
  105. package/dist/modes/interactive/session-share.js +8 -4
  106. package/dist/modes/interactive/tui-renderer.js +2 -2
  107. package/dist/modes/rpc/rpc-mode.js +2 -2
  108. package/dist/utils/clipboard-command.d.ts +6 -0
  109. package/dist/utils/clipboard-command.js +44 -0
  110. package/dist/utils/clipboard-image.js +55 -89
  111. package/dist/utils/clipboard.d.ts +1 -1
  112. package/dist/utils/clipboard.js +118 -108
  113. package/dist/utils/wsl.d.ts +2 -0
  114. package/dist/utils/wsl.js +14 -0
  115. package/dist/utils/zip.d.ts +6 -0
  116. package/dist/utils/zip.js +59 -0
  117. package/package.json +9 -9
  118. package/dist/utils/clipboard-native.d.ts +0 -10
  119. package/dist/utils/clipboard-native.js +0 -19
@@ -0,0 +1,341 @@
1
+ import { calculateCost, } from "@earendil-works/pi-ai";
2
+ import { getProviderEnvValue } from "@earendil-works/pi-ai/utils/provider-env";
3
+ /** Streaming warming never continues past this long after the real request that started it. */
4
+ const MAX_WARMING_AGE_MS = 60 * 60_000;
5
+ /** Idle warming uses a shorter horizon because continuation estimates become less reliable with age. */
6
+ const MAX_IDLE_WARMING_AGE_MS = 30 * 60_000;
7
+ /** A refresh is sent only when it is expected to save at least this many dollars. */
8
+ const CACHE_WARMING_MINIMUM_EXPECTED_SAVINGS = 0.05;
9
+ /**
10
+ * Chance that a real request arrives before the cache entry expires while the
11
+ * agent sits idle. Measured from our own usage; per-session estimates were not
12
+ * better than this constant.
13
+ */
14
+ const IDLE_CONTINUATION_PROBABILITY = 0.15;
15
+ /** Refresh at 90% of the TTL while preserving at least ten seconds of margin. */
16
+ export function getCacheWarmingDelayMs(ttlMs) {
17
+ if (ttlMs <= 10_000)
18
+ return undefined;
19
+ return Math.max(1, Math.floor(Math.min(ttlMs * 0.9, ttlMs - 10_000)));
20
+ }
21
+ /**
22
+ * Lifetime of the prompt cache entry a request writes, from the model's
23
+ * `promptCache` tier for the retention the request used. Undefined when the
24
+ * model has no lifetime for that tier or caching is off.
25
+ */
26
+ export function getPromptCacheTtlMs(model, options) {
27
+ const retention = options?.cacheRetention ??
28
+ (getProviderEnvValue("PI_CACHE_RETENTION", options?.env) === "long" ? "long" : "short");
29
+ if (retention === "none")
30
+ return undefined;
31
+ const seconds = model.promptCache?.[retention];
32
+ return seconds === undefined ? undefined : seconds * 1000;
33
+ }
34
+ /**
35
+ * Whether replaying the request with a one-token output cap leaves its cache
36
+ * entry untouched. Anthropic's budget-based thinking (Claude models without
37
+ * adaptive thinking) derives `budget_tokens` from `max_tokens`; the replay
38
+ * would get a different budget, which Anthropic keys the message cache on,
39
+ * and the model could still think for thousands of tokens.
40
+ */
41
+ export function isReplayable(model, options) {
42
+ if (!options?.reasoning || model.api !== "anthropic-messages")
43
+ return true;
44
+ return model.compat?.forceAdaptiveThinking === true;
45
+ }
46
+ /** Prompt size of the most recent real request on the branch, as reported by the provider. */
47
+ function lastPromptTokens(entries) {
48
+ for (let index = entries.length - 1; index >= 0; index--) {
49
+ const entry = entries[index];
50
+ if (entry.type === "message" && entry.message.role === "assistant") {
51
+ const usage = entry.message.usage;
52
+ return usage.input + usage.cacheRead + usage.cacheWrite;
53
+ }
54
+ }
55
+ return 0;
56
+ }
57
+ function price(model, tokens) {
58
+ const usage = {
59
+ input: 0,
60
+ output: 0,
61
+ cacheRead: 0,
62
+ cacheWrite: 0,
63
+ totalTokens: 0,
64
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
65
+ ...tokens,
66
+ };
67
+ return calculateCost(model, usage).total;
68
+ }
69
+ /**
70
+ * Keeps one prompt cache entry alive by re-sending its request with a
71
+ * one-token output cap before the entry expires. `start` replaces any
72
+ * previous run; warm requests never extend the fixed safety windows.
73
+ */
74
+ export class CacheWarmer {
75
+ run;
76
+ inactive;
77
+ models;
78
+ sessionManager;
79
+ getMode;
80
+ /** Lets extensions override `event.action`; failures fall back to pi's decision. */
81
+ decide;
82
+ /** Called with the persisted usage entry after each successful refresh. */
83
+ onWarmed;
84
+ constructor(models, sessionManager, getMode, decide = async (event) => event.action) {
85
+ this.models = models;
86
+ this.sessionManager = sessionManager;
87
+ this.getMode = getMode;
88
+ this.decide = decide;
89
+ this.inactive = { state: "inactive", reason: "waiting for first request" };
90
+ }
91
+ get status() {
92
+ if (this.getMode() === "off")
93
+ return { state: "inactive", reason: "cache warming disabled" };
94
+ const run = this.run;
95
+ if (!run)
96
+ return this.inactive;
97
+ if (!run.isCurrent())
98
+ return { state: "inactive", reason: "conversation context changed" };
99
+ const decision = this.evaluate(run);
100
+ const refreshing = run.timer === undefined;
101
+ if (!decision.economicsAvailable && !refreshing) {
102
+ return { state: "inactive", reason: "cache economics unavailable" };
103
+ }
104
+ return {
105
+ state: refreshing ? "refreshing" : "scheduled",
106
+ nextWarmAt: run.nextWarmAt,
107
+ decision,
108
+ extensionOverride: run.extensionOverride,
109
+ };
110
+ }
111
+ /** Keep the prompt cache entry written by `request` warm while `isCurrent` holds. */
112
+ start(request, isCurrent) {
113
+ this.clearRun();
114
+ const mode = this.getMode();
115
+ if (mode === "off") {
116
+ this.stop("cache warming disabled");
117
+ return;
118
+ }
119
+ if (!isReplayable(request.model, request.options)) {
120
+ this.stop("request cannot be replayed safely");
121
+ return;
122
+ }
123
+ const ttlMs = getPromptCacheTtlMs(request.model, request.options);
124
+ if (ttlMs === undefined) {
125
+ this.stop(request.options.cacheRetention === "none"
126
+ ? "request disabled prompt caching"
127
+ : "cache lifetime unavailable");
128
+ return;
129
+ }
130
+ const delayMs = getCacheWarmingDelayMs(ttlMs);
131
+ if (delayMs === undefined) {
132
+ this.stop("cache lifetime unavailable");
133
+ return;
134
+ }
135
+ this.run = {
136
+ ...request,
137
+ isCurrent,
138
+ delayMs,
139
+ startedAt: Date.now(),
140
+ controller: new AbortController(),
141
+ phase: "streaming",
142
+ nextWarmAt: 0,
143
+ extensionOverride: false,
144
+ };
145
+ this.schedule(this.run);
146
+ }
147
+ onAgentSettled() {
148
+ const run = this.run;
149
+ if (!run)
150
+ return;
151
+ if (this.getMode() === "streaming") {
152
+ this.stop("agent run settled");
153
+ return;
154
+ }
155
+ run.phase = "idle";
156
+ const deadline = run.startedAt + MAX_IDLE_WARMING_AGE_MS;
157
+ if (run.nextWarmAt > deadline || Date.now() >= deadline) {
158
+ this.stop("30-minute idle safety limit reached");
159
+ }
160
+ }
161
+ /** Reconcile an active run after the persisted warming mode changes. */
162
+ onModeChanged() {
163
+ const run = this.run;
164
+ if (!run)
165
+ return;
166
+ const reason = this.getModeStopReason(run);
167
+ if (reason)
168
+ this.stop(reason);
169
+ }
170
+ cancel() {
171
+ this.stop("inactive");
172
+ }
173
+ clearRun() {
174
+ const run = this.run;
175
+ if (!run)
176
+ return;
177
+ this.run = undefined;
178
+ if (run.timer)
179
+ clearTimeout(run.timer);
180
+ run.controller.abort();
181
+ }
182
+ stop(reason, stopped) {
183
+ this.clearRun();
184
+ this.inactive = { state: "inactive", reason, ...stopped };
185
+ }
186
+ schedule(run) {
187
+ run.extensionOverride = false;
188
+ run.nextWarmAt = Date.now() + run.delayMs;
189
+ const deadline = run.startedAt + (run.phase === "idle" ? MAX_IDLE_WARMING_AGE_MS : MAX_WARMING_AGE_MS);
190
+ if (run.nextWarmAt > deadline || Date.now() >= deadline) {
191
+ this.stop(run.phase === "idle" ? "30-minute idle safety limit reached" : "one-hour safety limit reached");
192
+ return;
193
+ }
194
+ run.timer = setTimeout(() => void this.refresh(run), Math.max(0, run.nextWarmAt - Date.now()));
195
+ run.timer.unref?.();
196
+ }
197
+ async refresh(run) {
198
+ run.timer = undefined;
199
+ if (!this.validateRun(run))
200
+ return;
201
+ const decision = this.evaluate(run);
202
+ const { warmCost, missCost, continuationProbability } = decision;
203
+ let action = decision.action;
204
+ try {
205
+ action = await this.decide({
206
+ type: "cache_warming_decision",
207
+ warmCost,
208
+ missCost,
209
+ continuationProbability,
210
+ action,
211
+ });
212
+ }
213
+ catch {
214
+ // Extension failures fall back to pi's own decision.
215
+ }
216
+ if (!this.validateRun(run))
217
+ return;
218
+ const extensionOverride = action !== decision.action;
219
+ if (action === "stop") {
220
+ const reason = extensionOverride
221
+ ? "stopped by extension"
222
+ : decision.economicsAvailable
223
+ ? "expected savings below threshold"
224
+ : "cache economics unavailable";
225
+ this.stop(reason, { decision, extensionOverride });
226
+ return;
227
+ }
228
+ run.extensionOverride = extensionOverride;
229
+ try {
230
+ const message = await this.models
231
+ .streamSimple(run.model, run.context, {
232
+ ...run.options,
233
+ maxTokens: 1,
234
+ maxRetries: 0,
235
+ signal: run.controller.signal,
236
+ })
237
+ .result();
238
+ if (!this.validateRun(run))
239
+ return;
240
+ if (message.stopReason !== "error" && message.stopReason !== "aborted") {
241
+ const entry = this.sessionManager.appendUsage("cache_warm", message.provider, message.responseModel ?? message.model, message.usage, extensionOverride ? "extension override" : undefined);
242
+ this.onWarmed?.(entry);
243
+ }
244
+ }
245
+ catch {
246
+ // Cache warming is best-effort and must not affect the active agent run.
247
+ }
248
+ if (this.run === run)
249
+ this.schedule(run);
250
+ }
251
+ validateRun(run) {
252
+ if (this.run !== run)
253
+ return false;
254
+ const reason = this.getModeStopReason(run) ?? (!run.isCurrent() ? "conversation context changed" : undefined);
255
+ if (!reason)
256
+ return true;
257
+ this.stop(reason);
258
+ return false;
259
+ }
260
+ getModeStopReason(run) {
261
+ const mode = this.getMode();
262
+ if (mode === "off")
263
+ return "cache warming disabled";
264
+ if (mode === "streaming" && run.phase === "idle")
265
+ return "agent run settled";
266
+ return undefined;
267
+ }
268
+ evaluate(run) {
269
+ const model = run.model;
270
+ const promptTokens = lastPromptTokens(this.sessionManager.getBranch());
271
+ const cacheHitCost = price(model, { cacheRead: promptTokens });
272
+ const cacheMissCost = price(model, model.cost.cacheWrite > 0 ? { cacheWrite: promptTokens } : { input: promptTokens });
273
+ const warmCost = price(model, { cacheRead: promptTokens, output: 1 });
274
+ const missCost = Math.max(0, cacheMissCost - cacheHitCost);
275
+ const continuationProbability = run.phase === "idle" ? IDLE_CONTINUATION_PROBABILITY : 1;
276
+ const economicsAvailable = promptTokens > 0 && (cacheHitCost > 0 || cacheMissCost > 0);
277
+ const expectedSavings = continuationProbability * missCost - warmCost;
278
+ return {
279
+ phase: run.phase,
280
+ warmCost,
281
+ missCost,
282
+ continuationProbability,
283
+ expectedSavings,
284
+ economicsAvailable,
285
+ action: expectedSavings >= CACHE_WARMING_MINIMUM_EXPECTED_SAVINGS ? "warm" : "stop",
286
+ };
287
+ }
288
+ }
289
+ function formatDollars(value) {
290
+ return value < 0 ? `-$${Math.abs(value).toFixed(3)}` : `$${value.toFixed(3)}`;
291
+ }
292
+ function formatCacheWarmingEconomics(decision) {
293
+ if (!decision.economicsAvailable)
294
+ return "cache economics unavailable";
295
+ const probability = Math.round(decision.continuationProbability * 100);
296
+ const probabilityText = decision.phase === "streaming"
297
+ ? `${probability}% continuation probability while agent is running`
298
+ : `${probability}% continuation probability`;
299
+ const comparison = decision.action === "warm" ? ">=" : "<";
300
+ return `${probabilityText}, expected savings ${formatDollars(decision.expectedSavings)} ${comparison} $${CACHE_WARMING_MINIMUM_EXPECTED_SAVINGS.toFixed(3)}`;
301
+ }
302
+ function formatCacheWarmingDecisionTime(nextWarmAt, now) {
303
+ if (nextWarmAt === undefined || nextWarmAt <= now)
304
+ return "Decision now";
305
+ let remainingSeconds = Math.ceil((nextWarmAt - now) / 1000);
306
+ const hours = Math.floor(remainingSeconds / 3600);
307
+ remainingSeconds %= 3600;
308
+ const minutes = Math.floor(remainingSeconds / 60);
309
+ const seconds = remainingSeconds % 60;
310
+ const parts = [];
311
+ if (hours > 0)
312
+ parts.push(`${hours}h`);
313
+ if (minutes > 0)
314
+ parts.push(`${minutes}m`);
315
+ if (seconds > 0 || parts.length === 0)
316
+ parts.push(`${seconds}s`);
317
+ return `Decision in ${parts.join(" ")}`;
318
+ }
319
+ /** One-line status for `/session`. */
320
+ export function formatCacheWarmingStatus(status, now = Date.now()) {
321
+ const decision = status.decision;
322
+ // A decision is attached once pi (or an extension) acted on it; "inactive"
323
+ // without one never got that far.
324
+ if (!decision || (status.state === "inactive" && !decision.economicsAvailable && !status.extensionOverride)) {
325
+ return `Inactive (${status.reason ?? "unknown reason"})`;
326
+ }
327
+ const details = status.extensionOverride
328
+ ? `extension override, ${formatCacheWarmingEconomics(decision)}`
329
+ : `${formatCacheWarmingEconomics(decision)} -> ${decision.action}`;
330
+ if (status.state === "inactive")
331
+ return `Stopped (${details})`;
332
+ if (status.state === "refreshing")
333
+ return `Warming cache (${details})`;
334
+ return `${formatCacheWarmingDecisionTime(status.nextWarmAt, now)} (${details})`;
335
+ }
336
+ /** One-line transcript text for persisted cache-warming usage. */
337
+ export function formatCacheWarmingUsage(entry) {
338
+ const note = entry.note ? ` (${entry.note})` : "";
339
+ const cost = entry.usage.cost.total.toFixed(6).replace(/(\.\d{3}\d*?)0+$/, "$1");
340
+ return `Cache warmed${note}: $${cost}`;
341
+ }
@@ -4,7 +4,7 @@
4
4
  * When navigating to a different point in the session tree, this generates
5
5
  * a summary of the branch being left so context isn't lost.
6
6
  */
7
- import { contentText } from "@earendil-works/pi-ai";
7
+ import { contentText, normalizeContext } from "@earendil-works/pi-ai";
8
8
  import { convertToLlm, createBranchSummaryMessage, createCompactionSummaryMessage, createCustomMessage, } from "../messages.js";
9
9
  import { completeSummarization, estimateTokens, getSummarizationFailure } from "./compaction.js";
10
10
  import { computeFileLists, createFileOps, extractFileOpsFromMessage, formatFileOperations, SUMMARIZATION_SYSTEM_PROMPT, serializeConversation, } from "./utils.js";
@@ -221,7 +221,7 @@ export async function generateBranchSummary(entries, options) {
221
221
  // request behavior (timeouts, retries, attribution headers) stays consistent
222
222
  // without running through agent state/events. Retried via completeSummarization
223
223
  // so transient stream drops reuse the configured retry policy.
224
- const context = { systemPrompt: SUMMARIZATION_SYSTEM_PROMPT, messages: summarizationMessages };
224
+ const context = normalizeContext({ systemPrompt: SUMMARIZATION_SYSTEM_PROMPT, messages: summarizationMessages });
225
225
  const requestOptions = { apiKey, headers, env, signal, maxTokens };
226
226
  const response = await completeSummarization(model, context, requestOptions, streamFn, retry, callbacks);
227
227
  // Check if aborted or errored
@@ -6,7 +6,7 @@
6
6
  */
7
7
  import type { AgentMessage, StreamFn, ThinkingLevel } from "@earendil-works/pi-agent-core";
8
8
  import { type RetryCallbacks, type RetryPolicy } from "@earendil-works/pi-ai";
9
- import type { AssistantMessage, Context, Model, SimpleStreamOptions, Usage } from "@earendil-works/pi-ai/compat";
9
+ import type { AssistantMessage, Model, SimpleStreamOptions, TranscriptContext, Usage } from "@earendil-works/pi-ai/compat";
10
10
  import { type SessionEntry } from "../session-manager.ts";
11
11
  import { type FileOperations } from "./utils.ts";
12
12
  /** Details stored in CompactionEntry.details for file tracking */
@@ -102,14 +102,14 @@ export declare function getSummarizationFailure(response: AssistantMessage, labe
102
102
  * the whole compaction on the first attempt. Deterministic errors and aborts return
103
103
  * immediately (see {@link retryAssistantCall}).
104
104
  */
105
- export declare function completeSummarization(model: Model<any>, context: Context, options: SimpleStreamOptions, streamFn?: StreamFn, retry?: RetryPolicy, callbacks?: RetryCallbacks): Promise<AssistantMessage>;
105
+ export declare function completeSummarization(model: Model<any>, context: TranscriptContext, options: SimpleStreamOptions, streamFn?: StreamFn, retry?: RetryPolicy, callbacks?: RetryCallbacks): Promise<AssistantMessage>;
106
106
  /**
107
107
  * Generate a summary of the conversation using the LLM.
108
108
  * If previousSummary is provided, uses the update prompt to merge.
109
109
  */
110
- export declare function generateSummary(currentMessages: AgentMessage[], model: Model<any>, reserveTokens: number, apiKey: string | undefined, headers?: Record<string, string>, signal?: AbortSignal, customInstructions?: string, previousSummary?: string, thinkingLevel?: ThinkingLevel, streamFn?: StreamFn, env?: Record<string, string>, retry?: RetryPolicy, callbacks?: RetryCallbacks): Promise<string>;
110
+ export declare function generateSummary(currentMessages: AgentMessage[], model: Model<any>, reserveTokens: number, apiKey: string | undefined, headers?: Record<string, string>, signal?: AbortSignal, customInstructions?: string, previousSummary?: string, thinkingLevel?: ThinkingLevel, streamFn?: StreamFn, env?: Record<string, string>, retry?: RetryPolicy, callbacks?: RetryCallbacks, sessionId?: string): Promise<string>;
111
111
  /** Generate or update a conversation summary and return its provider usage. */
112
- export declare function generateSummaryWithUsage(currentMessages: AgentMessage[], model: Model<any>, reserveTokens: number, apiKey: string | undefined, headers?: Record<string, string>, signal?: AbortSignal, customInstructions?: string, previousSummary?: string, thinkingLevel?: ThinkingLevel, streamFn?: StreamFn, env?: Record<string, string>, retry?: RetryPolicy, callbacks?: RetryCallbacks): Promise<{
112
+ export declare function generateSummaryWithUsage(currentMessages: AgentMessage[], model: Model<any>, reserveTokens: number, apiKey: string | undefined, headers?: Record<string, string>, signal?: AbortSignal, customInstructions?: string, previousSummary?: string, thinkingLevel?: ThinkingLevel, streamFn?: StreamFn, env?: Record<string, string>, retry?: RetryPolicy, callbacks?: RetryCallbacks, sessionId?: string): Promise<{
113
113
  text: string;
114
114
  usage: Usage;
115
115
  }>;
@@ -137,5 +137,6 @@ export declare function prepareCompaction(pathEntries: SessionEntry[], settings:
137
137
  *
138
138
  * @param preparation - Pre-calculated preparation from prepareCompaction()
139
139
  * @param customInstructions - Optional custom focus for the summary
140
+ * @param sessionId - Optional routing session ID forwarded without enabling prompt caching
140
141
  */
141
- export declare function compact(preparation: CompactionPreparation, model: Model<any>, apiKey: string | undefined, headers?: Record<string, string>, customInstructions?: string, signal?: AbortSignal, thinkingLevel?: ThinkingLevel, streamFn?: StreamFn, env?: Record<string, string>, retry?: RetryPolicy, callbacks?: RetryCallbacks): Promise<CompactionResult>;
142
+ export declare function compact(preparation: CompactionPreparation, model: Model<any>, apiKey: string | undefined, headers?: Record<string, string>, customInstructions?: string, signal?: AbortSignal, thinkingLevel?: ThinkingLevel, streamFn?: StreamFn, env?: Record<string, string>, retry?: RetryPolicy, callbacks?: RetryCallbacks, sessionId?: string): Promise<CompactionResult>;
@@ -4,7 +4,7 @@
4
4
  * Pure functions for compaction logic. The session manager handles I/O,
5
5
  * and after compaction the session is reloaded.
6
6
  */
7
- import { contentText, retryAssistantCall, uuidv7 } from "@earendil-works/pi-ai";
7
+ import { contentText, normalizeContext, retryAssistantCall, uuidv7, } from "@earendil-works/pi-ai";
8
8
  import { completeSimple } from "@earendil-works/pi-ai/compat";
9
9
  import { convertToLlm } from "../messages.js";
10
10
  import { buildSessionContext, sessionEntryToContextMessages, } from "../session-manager.js";
@@ -47,7 +47,9 @@ function getMessageFromEntryForCompaction(entry) {
47
47
  if (entry.type === "compaction") {
48
48
  return undefined;
49
49
  }
50
- return sessionEntryToContextMessages(entry)[0];
50
+ // System messages are prompt state, not conversation; the compaction entry carries their replay.
51
+ const message = sessionEntryToContextMessages(entry)[0];
52
+ return message?.role === "system" ? undefined : message;
51
53
  }
52
54
  function combineUsage(first, second) {
53
55
  return {
@@ -321,13 +323,10 @@ export function findCutPoint(entries, startIndex, endIndex, keepRecentTokens) {
321
323
  accumulatedTokens += messageTokens;
322
324
  // Check if we've exceeded the budget
323
325
  if (accumulatedTokens >= keepRecentTokens) {
324
- // Find the closest valid cut point at or after this entry
325
- for (let c = 0; c < cutPoints.length; c++) {
326
- if (cutPoints[c] >= i) {
327
- cutIndex = cutPoints[c];
328
- break;
329
- }
330
- }
326
+ // Prefer the closest valid cut point at or after this entry. If trailing
327
+ // tool results exceed the budget by themselves, keep their preceding
328
+ // assistant tool call instead of falling back to the first message.
329
+ cutIndex = cutPoints.find((candidate) => candidate >= i) ?? cutPoints[cutPoints.length - 1];
331
330
  break;
332
331
  }
333
332
  }
@@ -385,9 +384,7 @@ Use this EXACT format:
385
384
  - [Or "(none)" if not applicable]
386
385
 
387
386
  Keep each section concise. Preserve exact file paths, function names, and error messages.`;
388
- const UPDATE_SUMMARIZATION_PROMPT = `The messages above are NEW conversation messages to incorporate into the existing summary provided in <previous-summary> tags.
389
-
390
- Update the existing structured summary with new information. RULES:
387
+ const UPDATE_SUMMARIZATION_INSTRUCTIONS = `Update the existing structured summary with new information. RULES:
391
388
  - PRESERVE all existing information from the previous summary
392
389
  - ADD new progress, decisions, and context from the new messages
393
390
  - UPDATE the Progress section: move items from "In Progress" to "Done" when completed
@@ -423,6 +420,9 @@ Use this EXACT format:
423
420
  - [Preserve important context, add new if needed]
424
421
 
425
422
  Keep each section concise. Preserve exact file paths, function names, and error messages.`;
423
+ const UPDATE_SUMMARIZATION_PROMPT = `The messages above are NEW conversation messages to incorporate into the existing summary provided in <previous-summary> tags.
424
+
425
+ ${UPDATE_SUMMARIZATION_INSTRUCTIONS}`;
426
426
  /**
427
427
  * Returns an error message when a summarization response cannot safely be persisted.
428
428
  * A length stop contains partial text and must not become a session checkpoint.
@@ -436,8 +436,8 @@ export function getSummarizationFailure(response, label) {
436
436
  }
437
437
  return undefined;
438
438
  }
439
- function createSummarizationOptions(model, maxTokens, apiKey, headers, env, signal, thinkingLevel) {
440
- const options = { maxTokens, signal, apiKey, headers, env };
439
+ function createSummarizationOptions(model, maxTokens, apiKey, headers, env, signal, thinkingLevel, sessionId) {
440
+ const options = { maxTokens, signal, apiKey, headers, env, sessionId };
441
441
  if (model.reasoning && thinkingLevel && thinkingLevel !== "off") {
442
442
  options.reasoning = thinkingLevel;
443
443
  }
@@ -451,11 +451,12 @@ function createSummarizationOptions(model, maxTokens, apiKey, headers, env, sign
451
451
  * immediately (see {@link retryAssistantCall}).
452
452
  */
453
453
  export async function completeSummarization(model, context, options, streamFn, retry, callbacks) {
454
- // Summaries are standalone requests, so isolate routing and avoid cache writes that cannot be reused.
454
+ // Avoid cache writes for one-off summaries. Reuse caller-supplied routing when available;
455
+ // callers without a session ID, including branch summaries, receive a fresh routing ID.
455
456
  const requestOptions = {
456
457
  ...options,
457
458
  cacheRetention: "none",
458
- sessionId: uuidv7(),
459
+ sessionId: options.sessionId ?? uuidv7(),
459
460
  };
460
461
  const produce = async () => streamFn
461
462
  ? (await streamFn(model, context, requestOptions)).result()
@@ -466,11 +467,24 @@ export async function completeSummarization(model, context, options, streamFn, r
466
467
  * Generate a summary of the conversation using the LLM.
467
468
  * If previousSummary is provided, uses the update prompt to merge.
468
469
  */
469
- export async function generateSummary(currentMessages, model, reserveTokens, apiKey, headers, signal, customInstructions, previousSummary, thinkingLevel, streamFn, env, retry, callbacks) {
470
- return (await generateSummaryWithUsage(currentMessages, model, reserveTokens, apiKey, headers, signal, customInstructions, previousSummary, thinkingLevel, streamFn, env, retry, callbacks)).text;
470
+ export async function generateSummary(currentMessages, model, reserveTokens, apiKey, headers, signal, customInstructions, previousSummary, thinkingLevel, streamFn, env, retry, callbacks, sessionId) {
471
+ return (await generateSummaryWithUsage(currentMessages, model, reserveTokens, apiKey, headers, signal, customInstructions, previousSummary, thinkingLevel, streamFn, env, retry, callbacks, sessionId)).text;
472
+ }
473
+ /** Build the provider context for a standalone summary request. */
474
+ function buildSummarizationContext(promptText) {
475
+ return normalizeContext({
476
+ systemPrompt: SUMMARIZATION_SYSTEM_PROMPT,
477
+ messages: [
478
+ {
479
+ role: "user",
480
+ content: [{ type: "text", text: promptText }],
481
+ timestamp: Date.now(),
482
+ },
483
+ ],
484
+ });
471
485
  }
472
486
  /** Generate or update a conversation summary and return its provider usage. */
473
- export async function generateSummaryWithUsage(currentMessages, model, reserveTokens, apiKey, headers, signal, customInstructions, previousSummary, thinkingLevel, streamFn, env, retry, callbacks) {
487
+ export async function generateSummaryWithUsage(currentMessages, model, reserveTokens, apiKey, headers, signal, customInstructions, previousSummary, thinkingLevel, streamFn, env, retry, callbacks, sessionId) {
474
488
  const maxTokens = Math.min(Math.floor(0.8 * reserveTokens), model.maxTokens > 0 ? model.maxTokens : Number.POSITIVE_INFINITY);
475
489
  // Use update prompt if we have a previous summary, otherwise initial prompt
476
490
  let basePrompt = previousSummary ? UPDATE_SUMMARIZATION_PROMPT : SUMMARIZATION_PROMPT;
@@ -487,19 +501,15 @@ export async function generateSummaryWithUsage(currentMessages, model, reserveTo
487
501
  promptText += `<previous-summary>\n${previousSummary}\n</previous-summary>\n\n`;
488
502
  }
489
503
  promptText += basePrompt;
490
- const summarizationMessages = [
491
- {
492
- role: "user",
493
- content: [{ type: "text", text: promptText }],
494
- timestamp: Date.now(),
495
- },
496
- ];
497
- const completionOptions = createSummarizationOptions(model, maxTokens, apiKey, headers, env, signal, thinkingLevel);
498
- const response = await completeSummarization(model, { systemPrompt: SUMMARIZATION_SYSTEM_PROMPT, messages: summarizationMessages }, completionOptions, streamFn, retry, callbacks);
504
+ const completionOptions = createSummarizationOptions(model, maxTokens, apiKey, headers, env, signal, thinkingLevel, sessionId);
505
+ const response = await completeSummarization(model, buildSummarizationContext(promptText), completionOptions, streamFn, retry, callbacks);
499
506
  const failure = getSummarizationFailure(response, "Summarization");
500
507
  if (failure) {
501
508
  throw new Error(failure);
502
509
  }
510
+ if (response.content.some((block) => block.type === "toolCall")) {
511
+ throw new Error("Summarization attempted to call a tool");
512
+ }
503
513
  const textContent = contentText(response.content);
504
514
  return { text: textContent, usage: response.usage };
505
515
  }
@@ -593,28 +603,29 @@ Be concise. Focus on what's needed to understand the kept suffix.`;
593
603
  *
594
604
  * @param preparation - Pre-calculated preparation from prepareCompaction()
595
605
  * @param customInstructions - Optional custom focus for the summary
606
+ * @param sessionId - Optional routing session ID forwarded without enabling prompt caching
596
607
  */
597
- export async function compact(preparation, model, apiKey, headers, customInstructions, signal, thinkingLevel, streamFn, env, retry, callbacks) {
608
+ export async function compact(preparation, model, apiKey, headers, customInstructions, signal, thinkingLevel, streamFn, env, retry, callbacks, sessionId) {
598
609
  const { firstKeptEntryId, messagesToSummarize, turnPrefixMessages, isSplitTurn, tokensBefore, previousSummary, fileOps, settings, } = preparation;
599
610
  // Generate summaries and merge into one
600
611
  let summary;
601
612
  let summaryUsage;
602
613
  if (isSplitTurn && turnPrefixMessages.length > 0) {
603
- let historyText = "No prior history.";
614
+ let historyText = previousSummary ?? "No prior history.";
604
615
  let historyUsage;
605
616
  if (messagesToSummarize.length > 0) {
606
- const historyResult = await generateSummaryWithUsage(messagesToSummarize, model, settings.reserveTokens, apiKey, headers, signal, customInstructions, previousSummary, thinkingLevel, streamFn, env, retry, callbacks);
617
+ const historyResult = await generateSummaryWithUsage(messagesToSummarize, model, settings.reserveTokens, apiKey, headers, signal, customInstructions, previousSummary, thinkingLevel, streamFn, env, retry, callbacks, sessionId);
607
618
  historyText = historyResult.text;
608
619
  historyUsage = historyResult.usage;
609
620
  }
610
- const turnPrefixResult = await generateTurnPrefixSummary(turnPrefixMessages, model, settings.reserveTokens, apiKey, headers, env, signal, thinkingLevel, streamFn, retry, callbacks);
621
+ const turnPrefixResult = await generateTurnPrefixSummary(turnPrefixMessages, model, settings.reserveTokens, apiKey, headers, env, signal, thinkingLevel, streamFn, retry, callbacks, sessionId);
611
622
  // Merge into single summary
612
623
  summary = `${historyText}\n\n---\n\n**Turn Context (split turn):**\n\n${turnPrefixResult.text}`;
613
624
  summaryUsage = historyUsage ? combineUsage(historyUsage, turnPrefixResult.usage) : turnPrefixResult.usage;
614
625
  }
615
626
  else {
616
627
  // Just generate history summary
617
- const result = await generateSummaryWithUsage(messagesToSummarize, model, settings.reserveTokens, apiKey, headers, signal, customInstructions, previousSummary, thinkingLevel, streamFn, env, retry, callbacks);
628
+ const result = await generateSummaryWithUsage(messagesToSummarize, model, settings.reserveTokens, apiKey, headers, signal, customInstructions, previousSummary, thinkingLevel, streamFn, env, retry, callbacks, sessionId);
618
629
  summary = result.text;
619
630
  summaryUsage = result.usage;
620
631
  }
@@ -635,23 +646,19 @@ export async function compact(preparation, model, apiKey, headers, customInstruc
635
646
  /**
636
647
  * Generate a summary for a turn prefix (when splitting a turn).
637
648
  */
638
- async function generateTurnPrefixSummary(messages, model, reserveTokens, apiKey, headers, env, signal, thinkingLevel, streamFn, retry, callbacks) {
649
+ async function generateTurnPrefixSummary(messages, model, reserveTokens, apiKey, headers, env, signal, thinkingLevel, streamFn, retry, callbacks, sessionId) {
639
650
  const maxTokens = Math.min(Math.floor(0.5 * reserveTokens), model.maxTokens > 0 ? model.maxTokens : Number.POSITIVE_INFINITY); // Smaller budget for turn prefix
640
651
  const llmMessages = convertToLlm(messages);
641
652
  const conversationText = serializeConversation(llmMessages);
642
653
  const promptText = `<conversation>\n${conversationText}\n</conversation>\n\n${TURN_PREFIX_SUMMARIZATION_PROMPT}`;
643
- const summarizationMessages = [
644
- {
645
- role: "user",
646
- content: [{ type: "text", text: promptText }],
647
- timestamp: Date.now(),
648
- },
649
- ];
650
- const response = await completeSummarization(model, { systemPrompt: SUMMARIZATION_SYSTEM_PROMPT, messages: summarizationMessages }, createSummarizationOptions(model, maxTokens, apiKey, headers, env, signal, thinkingLevel), streamFn, retry, callbacks);
654
+ const response = await completeSummarization(model, buildSummarizationContext(promptText), createSummarizationOptions(model, maxTokens, apiKey, headers, env, signal, thinkingLevel, sessionId), streamFn, retry, callbacks);
651
655
  const failure = getSummarizationFailure(response, "Turn prefix summarization");
652
656
  if (failure) {
653
657
  throw new Error(failure);
654
658
  }
659
+ if (response.content.some((block) => block.type === "toolCall")) {
660
+ throw new Error("Turn prefix summarization attempted to call a tool");
661
+ }
655
662
  return {
656
663
  text: contentText(response.content),
657
664
  usage: response.usage,
@@ -0,0 +1,21 @@
1
+ export interface CrashRecord {
2
+ timestamp: string;
3
+ version: string;
4
+ kind: "uncaught_exception" | "fatal_error";
5
+ message: string;
6
+ stack: string | null;
7
+ sessionFile: string | null;
8
+ cwd: string;
9
+ notified?: boolean;
10
+ }
11
+ export declare function readCrashLog(path?: string): CrashRecord[];
12
+ /** Best-effort persistence for callers that are already crashing. */
13
+ export declare function recordCrash(crash: {
14
+ kind: CrashRecord["kind"];
15
+ error: unknown;
16
+ sessionFile?: string;
17
+ cwd: string;
18
+ }, path?: string): CrashRecord | undefined;
19
+ /** Return the newest recent crash, marking pending records as announced. */
20
+ export declare function takeUnnotifiedCrash(path?: string, now?: number): CrashRecord | undefined;
21
+ export declare function clearCrashLog(path?: string): void;