@oh-my-pi/pi-agent-core 17.3.7 → 17.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +27 -0
- package/dist/types/agent.d.ts +8 -1
- package/dist/types/compaction/branch-summarization.d.ts +2 -1
- package/dist/types/compaction/compaction.d.ts +37 -17
- package/dist/types/compaction/entries.d.ts +10 -1
- package/dist/types/compaction/index.d.ts +1 -0
- package/dist/types/compaction/message-cache.d.ts +6 -4
- package/dist/types/compaction/messages.d.ts +17 -1
- package/dist/types/compaction/openai.d.ts +2 -1
- package/dist/types/compaction/pruning.d.ts +3 -2
- package/dist/types/compaction/shake.d.ts +2 -1
- package/dist/types/compaction/transcript-tokens.d.ts +76 -0
- package/dist/types/compaction/utils.d.ts +2 -0
- package/dist/types/tokenizer.d.ts +78 -2
- package/dist/types/types.d.ts +1 -1
- package/package.json +8 -8
- package/src/agent.ts +25 -4
- package/src/compaction/branch-summarization.ts +15 -8
- package/src/compaction/compaction.ts +234 -162
- package/src/compaction/entries.ts +11 -0
- package/src/compaction/index.ts +1 -0
- package/src/compaction/message-cache.ts +33 -34
- package/src/compaction/messages.ts +21 -5
- package/src/compaction/openai.ts +52 -20
- package/src/compaction/prompts/summarization-system.md +2 -0
- package/src/compaction/pruning.ts +25 -13
- package/src/compaction/shake.ts +18 -16
- package/src/compaction/transcript-tokens.ts +111 -0
- package/src/compaction/utils.ts +9 -2
- package/src/tokenizer.ts +273 -18
- package/src/types.ts +1 -1
|
@@ -9,8 +9,8 @@ import type { Api, ApiKey, AssistantMessage, Context, Model, SimpleStreamOptions
|
|
|
9
9
|
import { preferredDialect } from "@oh-my-pi/pi-catalog/identity";
|
|
10
10
|
import { prompt } from "@oh-my-pi/pi-utils";
|
|
11
11
|
import { type AgentTelemetry, instrumentedCompleteSimple } from "../telemetry";
|
|
12
|
+
import { Tokenizer } from "../tokenizer";
|
|
12
13
|
import type { AgentMessage } from "../types";
|
|
13
|
-
import { estimateTokens } from "./compaction";
|
|
14
14
|
import type { ReadonlySessionManager, SessionEntry } from "./entries";
|
|
15
15
|
import {
|
|
16
16
|
type ConvertToLlm,
|
|
@@ -191,7 +191,9 @@ function getMessageFromEntry(entry: SessionEntry): AgentMessage | undefined {
|
|
|
191
191
|
return createBranchSummaryMessage(entry.summary, entry.fromId, entry.timestamp);
|
|
192
192
|
|
|
193
193
|
case "compaction":
|
|
194
|
-
return createCompactionSummaryMessage(entry.summary, entry.tokensBefore, entry.timestamp,
|
|
194
|
+
return createCompactionSummaryMessage(entry.summary, entry.tokensBefore, entry.timestamp, {
|
|
195
|
+
shortSummary: entry.shortSummary,
|
|
196
|
+
});
|
|
195
197
|
|
|
196
198
|
// These don't contribute to conversation content
|
|
197
199
|
case "thinking_level_change":
|
|
@@ -206,14 +208,14 @@ function getMessageFromEntry(entry: SessionEntry): AgentMessage | undefined {
|
|
|
206
208
|
}
|
|
207
209
|
}
|
|
208
210
|
|
|
209
|
-
function estimateBranchSummaryTokens(message: AgentMessage): number {
|
|
210
|
-
if (message.role !== "toolResult") return
|
|
211
|
+
function estimateBranchSummaryTokens(message: AgentMessage, tokenizer: Tokenizer): number {
|
|
212
|
+
if (message.role !== "toolResult") return tokenizer.countMessage(message);
|
|
211
213
|
const text = message.content
|
|
212
214
|
.filter((c): c is { type: "text"; text: string } => c.type === "text")
|
|
213
215
|
.map(c => c.text)
|
|
214
216
|
.join("");
|
|
215
217
|
if (!text) return 0;
|
|
216
|
-
return
|
|
218
|
+
return tokenizer.countMessage({
|
|
217
219
|
...message,
|
|
218
220
|
content: [{ type: "text", text: truncateToolResultForSummary(text) }],
|
|
219
221
|
});
|
|
@@ -232,7 +234,11 @@ function estimateBranchSummaryTokens(message: AgentMessage): number {
|
|
|
232
234
|
* @param entries - Entries in chronological order
|
|
233
235
|
* @param tokenBudget - Maximum tokens to include (0 = no limit)
|
|
234
236
|
*/
|
|
235
|
-
export function prepareBranchEntries(
|
|
237
|
+
export function prepareBranchEntries(
|
|
238
|
+
entries: SessionEntry[],
|
|
239
|
+
tokenizer: Tokenizer,
|
|
240
|
+
tokenBudget: number = 0,
|
|
241
|
+
): BranchPreparation {
|
|
236
242
|
const messages: AgentMessage[] = [];
|
|
237
243
|
const fileOps = createFileOps();
|
|
238
244
|
let totalTokens = 0;
|
|
@@ -264,7 +270,7 @@ export function prepareBranchEntries(entries: SessionEntry[], tokenBudget: numbe
|
|
|
264
270
|
// Extract file ops from assistant messages (tool calls)
|
|
265
271
|
extractFileOpsFromMessage(message, fileOps);
|
|
266
272
|
|
|
267
|
-
const tokens = estimateBranchSummaryTokens(message);
|
|
273
|
+
const tokens = estimateBranchSummaryTokens(message, tokenizer);
|
|
268
274
|
|
|
269
275
|
// Check budget before adding
|
|
270
276
|
if (tokenBudget > 0 && totalTokens + tokens > tokenBudget) {
|
|
@@ -309,8 +315,9 @@ export async function generateBranchSummary(
|
|
|
309
315
|
// Token budget = context window minus reserved space for prompt + response
|
|
310
316
|
const contextWindow = model.contextWindow || 128000;
|
|
311
317
|
const tokenBudget = contextWindow - reserveTokens;
|
|
318
|
+
const tokenizer = new Tokenizer(model);
|
|
312
319
|
|
|
313
|
-
const { messages, fileOps } = prepareBranchEntries(entries, tokenBudget);
|
|
320
|
+
const { messages, fileOps } = prepareBranchEntries(entries, tokenizer, tokenBudget);
|
|
314
321
|
|
|
315
322
|
if (messages.length === 0) {
|
|
316
323
|
return { summary: "No content to summarize" };
|
|
@@ -23,6 +23,7 @@ import {
|
|
|
23
23
|
type Usage,
|
|
24
24
|
withAuth,
|
|
25
25
|
} from "@oh-my-pi/pi-ai";
|
|
26
|
+
import type { Dialect } from "@oh-my-pi/pi-ai/dialect";
|
|
26
27
|
import * as AIError from "@oh-my-pi/pi-ai/error";
|
|
27
28
|
import { createOpenAICodexCompactionRequestContext } from "@oh-my-pi/pi-ai/providers/openai-codex-responses";
|
|
28
29
|
import { convertTools } from "@oh-my-pi/pi-ai/providers/openai-responses";
|
|
@@ -30,11 +31,11 @@ import { buildResponsesInput, resolveOpenAICompatPolicy } from "@oh-my-pi/pi-ai/
|
|
|
30
31
|
import { stripOpenAIResponsesOutputOnlyStatusesForReplay } from "@oh-my-pi/pi-ai/utils";
|
|
31
32
|
import { preferredDialect } from "@oh-my-pi/pi-catalog/identity";
|
|
32
33
|
import { clampThinkingLevelForModel } from "@oh-my-pi/pi-catalog/model-thinking";
|
|
33
|
-
import { isRecord, logger, prompt
|
|
34
|
+
import { isRecord, logger, prompt } from "@oh-my-pi/pi-utils";
|
|
34
35
|
import * as snapcompact from "@oh-my-pi/snapcompact";
|
|
35
36
|
import { type AgentTelemetry, instrumentedCompleteSimple } from "../telemetry";
|
|
36
37
|
import { ThinkingLevel } from "../thinking";
|
|
37
|
-
import {
|
|
38
|
+
import { Tokenizer } from "../tokenizer";
|
|
38
39
|
import type { AgentMessage } from "../types";
|
|
39
40
|
import {
|
|
40
41
|
buildCompactionV2Request,
|
|
@@ -46,7 +47,6 @@ import {
|
|
|
46
47
|
} from "./compaction-v2-streaming";
|
|
47
48
|
import type { CompactionEntry, SessionEntry } from "./entries";
|
|
48
49
|
import { NativeCompactionError } from "./errors";
|
|
49
|
-
import { isEstimateCacheable, readEstimateCache, writeEstimateCache } from "./message-cache";
|
|
50
50
|
import { type ConvertToLlm, createBranchSummaryMessage, createCustomMessage, defaultConvertToLlm } from "./messages";
|
|
51
51
|
import {
|
|
52
52
|
buildOpenAiNativeHistory,
|
|
@@ -68,6 +68,7 @@ import snapcompactArchiveContextPrompt from "./prompts/snapcompact-archive-conte
|
|
|
68
68
|
import {
|
|
69
69
|
computeFileLists,
|
|
70
70
|
createFileOps,
|
|
71
|
+
escapeSummaryBoundaryTags,
|
|
71
72
|
extractFileOpsFromMessage,
|
|
72
73
|
type FileOperations,
|
|
73
74
|
SUMMARIZATION_SYSTEM_PROMPT,
|
|
@@ -388,144 +389,17 @@ export function resolveThresholdTokens(contextWindow: number, settings: Compacti
|
|
|
388
389
|
// Cut point detection
|
|
389
390
|
// ============================================================================
|
|
390
391
|
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
/**
|
|
398
|
-
* Estimate token count for a message using cl100k_base via the native
|
|
399
|
-
* tokenizer. This is not Claude's first-party tokenizer (Anthropic doesn't
|
|
400
|
-
* publish one) but is within ~5–10% across English/code text.
|
|
401
|
-
*
|
|
402
|
-
* `excludeEncryptedReasoning` drops opaque provider reasoning payloads
|
|
403
|
-
* (`thinkingSignature`, `redactedThinking`) from the estimate. Those are billed
|
|
404
|
-
* by the provider on replay, so the default counts them — but their *local*
|
|
405
|
-
* byte size can diverge wildly from what the provider charges, so the
|
|
406
|
-
* compaction floor (which only needs the reliably-countable, on-wire-compressible
|
|
407
|
-
* content) excludes them to avoid false triggers on thinking-heavy turns.
|
|
408
|
-
*/
|
|
409
|
-
export function estimateTokens(message: AgentMessage, options?: { excludeEncryptedReasoning?: boolean }): number {
|
|
410
|
-
// Settled historical messages are counted once and reused until an owner
|
|
411
|
-
// (prune/shake/strip-images) invalidates them; streaming assistants bypass
|
|
412
|
-
// the cache entirely (see message-cache.ts settle-gate invariant).
|
|
413
|
-
const cacheable = isEstimateCacheable(message);
|
|
414
|
-
const excludeEncryptedReasoning = options?.excludeEncryptedReasoning === true;
|
|
415
|
-
if (cacheable) {
|
|
416
|
-
const cached = readEstimateCache(message, excludeEncryptedReasoning);
|
|
417
|
-
if (cached !== undefined) return cached;
|
|
418
|
-
}
|
|
419
|
-
const result = computeMessageTokens(message, options);
|
|
420
|
-
if (cacheable) writeEstimateCache(message, excludeEncryptedReasoning, result);
|
|
421
|
-
return result;
|
|
422
|
-
}
|
|
423
|
-
|
|
424
|
-
function computeMessageTokens(message: AgentMessage, options?: { excludeEncryptedReasoning?: boolean }): number {
|
|
425
|
-
const fragments: string[] = [];
|
|
426
|
-
let extra = 0;
|
|
427
|
-
if ((message as { role?: string }).role === "bashExecution") {
|
|
428
|
-
const bash = message as { command?: unknown; output?: unknown };
|
|
429
|
-
if (typeof bash.command === "string") fragments.push(bash.command);
|
|
430
|
-
if (typeof bash.output === "string") fragments.push(bash.output);
|
|
431
|
-
return fragments.length === 0 ? 0 : countTokens(fragments);
|
|
432
|
-
}
|
|
433
|
-
|
|
434
|
-
switch (message.role) {
|
|
435
|
-
case "user": {
|
|
436
|
-
const content = (message as { content: string | Array<{ type: string; text?: string }> }).content;
|
|
437
|
-
if (typeof content === "string") {
|
|
438
|
-
fragments.push(content);
|
|
439
|
-
} else if (Array.isArray(content)) {
|
|
440
|
-
for (const block of content) {
|
|
441
|
-
if (block.type === "text" && block.text) {
|
|
442
|
-
fragments.push(block.text);
|
|
443
|
-
}
|
|
444
|
-
}
|
|
445
|
-
}
|
|
446
|
-
break;
|
|
447
|
-
}
|
|
448
|
-
case "assistant": {
|
|
449
|
-
const assistant = message as AssistantMessage;
|
|
450
|
-
for (const block of assistant.content) {
|
|
451
|
-
if (block.type === "text") {
|
|
452
|
-
fragments.push(block.text);
|
|
453
|
-
} else if (block.type === "thinking") {
|
|
454
|
-
fragments.push(block.thinking);
|
|
455
|
-
// Providers charge for the opaque signature/reasoning payload that
|
|
456
|
-
// rides alongside the thinking text (OpenAI Responses encrypted
|
|
457
|
-
// reasoning items, Anthropic signed thinking blocks, etc.). Without
|
|
458
|
-
// counting it, this estimator can read ~half of the provider-reported
|
|
459
|
-
// usage on thinking-heavy turns — see #2275 for the resulting
|
|
460
|
-
// compaction-trigger / post-check metric divergence. The compaction
|
|
461
|
-
// floor excludes it (its local byte size diverges from provider billing).
|
|
462
|
-
if (block.thinkingSignature && !options?.excludeEncryptedReasoning) {
|
|
463
|
-
fragments.push(block.thinkingSignature);
|
|
464
|
-
}
|
|
465
|
-
} else if (block.type === "toolCall") {
|
|
466
|
-
fragments.push(block.name);
|
|
467
|
-
fragments.push(stringifyJson(block.arguments) ?? "null");
|
|
468
|
-
} else if (block.type === "redactedThinking") {
|
|
469
|
-
// Encrypted reasoning blob the provider still bills for on replay;
|
|
470
|
-
// excluded from the compaction floor for the same reason as above.
|
|
471
|
-
if (!options?.excludeEncryptedReasoning) fragments.push(block.data);
|
|
472
|
-
} else if (block.type === "anthropicServerTool") {
|
|
473
|
-
// Native Anthropic server-tool call/result replayed verbatim on the
|
|
474
|
-
// wire (server_tool_use input and opaque result content). This opaque
|
|
475
|
-
// provider-replay state the provider still
|
|
476
|
-
// bills for on same-provider replay; excluded from the compaction
|
|
477
|
-
// floor like other encrypted reasoning because its local byte size
|
|
478
|
-
// diverges from provider billing.
|
|
479
|
-
if (!options?.excludeEncryptedReasoning) fragments.push(stringifyJson(block.block) ?? "null");
|
|
480
|
-
}
|
|
481
|
-
}
|
|
482
|
-
break;
|
|
483
|
-
}
|
|
484
|
-
case "hookMessage":
|
|
485
|
-
case "toolResult": {
|
|
486
|
-
if (typeof message.content === "string") {
|
|
487
|
-
fragments.push(message.content);
|
|
488
|
-
} else {
|
|
489
|
-
for (const block of message.content) {
|
|
490
|
-
if (block.type === "text" && block.text) {
|
|
491
|
-
fragments.push(block.text);
|
|
492
|
-
} else if (block.type === "image") {
|
|
493
|
-
extra += IMAGE_TOKEN_ESTIMATE;
|
|
494
|
-
}
|
|
495
|
-
}
|
|
496
|
-
}
|
|
497
|
-
break;
|
|
498
|
-
}
|
|
499
|
-
case "branchSummary":
|
|
500
|
-
case "compactionSummary": {
|
|
501
|
-
fragments.push(message.summary);
|
|
502
|
-
if (message.role === "compactionSummary") {
|
|
503
|
-
if (message.blocks) {
|
|
504
|
-
for (const block of message.blocks) {
|
|
505
|
-
if (block.type === "text") fragments.push(block.text);
|
|
506
|
-
else extra += snapcompact.FRAME_TOKEN_ESTIMATE;
|
|
507
|
-
}
|
|
508
|
-
} else if (message.images) {
|
|
509
|
-
// Snapcompact frames render at ≥1568px; providers bill the downscaled cap.
|
|
510
|
-
extra += message.images.length * snapcompact.FRAME_TOKEN_ESTIMATE;
|
|
511
|
-
}
|
|
512
|
-
}
|
|
513
|
-
break;
|
|
514
|
-
}
|
|
515
|
-
default:
|
|
516
|
-
return 0;
|
|
517
|
-
}
|
|
518
|
-
|
|
519
|
-
if (fragments.length === 0) return extra;
|
|
520
|
-
return extra + countTokens(fragments);
|
|
521
|
-
}
|
|
522
|
-
|
|
523
|
-
function estimateEntriesTokens(entries: SessionEntry[], startIndex: number, endIndex: number): number {
|
|
392
|
+
function estimateEntriesTokens(
|
|
393
|
+
entries: SessionEntry[],
|
|
394
|
+
tokenizer: Tokenizer,
|
|
395
|
+
startIndex: number,
|
|
396
|
+
endIndex: number,
|
|
397
|
+
): number {
|
|
524
398
|
let total = 0;
|
|
525
399
|
for (let i = startIndex; i < endIndex; i++) {
|
|
526
400
|
const msg = getMessageFromEntry(entries[i]);
|
|
527
401
|
if (msg) {
|
|
528
|
-
total +=
|
|
402
|
+
total += tokenizer.countMessage(msg);
|
|
529
403
|
}
|
|
530
404
|
}
|
|
531
405
|
return total;
|
|
@@ -624,6 +498,7 @@ export interface CutPointResult {
|
|
|
624
498
|
*/
|
|
625
499
|
export function findCutPoint(
|
|
626
500
|
entries: SessionEntry[],
|
|
501
|
+
tokenizer: Tokenizer,
|
|
627
502
|
startIndex: number,
|
|
628
503
|
endIndex: number,
|
|
629
504
|
keepRecentTokens: number,
|
|
@@ -643,7 +518,7 @@ export function findCutPoint(
|
|
|
643
518
|
if (entry.type !== "message") continue;
|
|
644
519
|
|
|
645
520
|
// Estimate this message's size
|
|
646
|
-
const messageTokens =
|
|
521
|
+
const messageTokens = tokenizer.countMessage(entry.message);
|
|
647
522
|
accumulatedTokens += messageTokens;
|
|
648
523
|
|
|
649
524
|
// Check if we've exceeded the budget
|
|
@@ -887,6 +762,88 @@ function createSnapcompactArchiveMigrationMessage(archiveText: string): Message
|
|
|
887
762
|
};
|
|
888
763
|
}
|
|
889
764
|
|
|
765
|
+
/**
|
|
766
|
+
* Fallback window for a model whose catalog entry carries no usable context
|
|
767
|
+
* window; matches the smallest window any compaction-capable model ships with.
|
|
768
|
+
*/
|
|
769
|
+
const DEFAULT_SUMMARY_INPUT_WINDOW = 200_000;
|
|
770
|
+
|
|
771
|
+
/**
|
|
772
|
+
* Floor for one summarization window, so a tiny model still makes progress.
|
|
773
|
+
* Scaled down (never below 1k) for models whose window cannot host the full
|
|
774
|
+
* floor next to the carried summary and output reserves.
|
|
775
|
+
*/
|
|
776
|
+
const MIN_SUMMARY_INPUT_TOKENS = 16_384;
|
|
777
|
+
|
|
778
|
+
/** Smallest window worth planning for `model`; below this, overflow recovery gives up. */
|
|
779
|
+
function minSummaryInputTokens(model: Model): number {
|
|
780
|
+
const window = model.contextWindow && model.contextWindow > 0 ? model.contextWindow : DEFAULT_SUMMARY_INPUT_WINDOW;
|
|
781
|
+
return Math.min(MIN_SUMMARY_INPUT_TOKENS, Math.max(1_024, Math.floor(window / 8)));
|
|
782
|
+
}
|
|
783
|
+
|
|
784
|
+
/**
|
|
785
|
+
* Usable conversation input for ONE summarization call: the summarizer's window
|
|
786
|
+
* minus the summary it must emit, the previous summary it carries forward, and
|
|
787
|
+
* prompt scaffolding. Providers tokenize differently from the local cl100k
|
|
788
|
+
* estimate, so the window is discounted before the fixed reserves come off.
|
|
789
|
+
*/
|
|
790
|
+
function summaryInputBudgetTokens(model: Model, maxTokens: number): number {
|
|
791
|
+
const window = model.contextWindow && model.contextWindow > 0 ? model.contextWindow : DEFAULT_SUMMARY_INPUT_WINDOW;
|
|
792
|
+
// 0.8, not "window minus reserves": provider tokenizers disagree with the
|
|
793
|
+
// local cl100k estimate by a few percent, and being wrong here is a hard
|
|
794
|
+
// 400 on the one call that is supposed to rescue an oversized session.
|
|
795
|
+
return Math.max(minSummaryInputTokens(model), Math.floor(window * 0.8) - maxTokens - MAX_SUMMARY_TOKENS);
|
|
796
|
+
}
|
|
797
|
+
|
|
798
|
+
/**
|
|
799
|
+
* Clamp one serialized window to the budget. Only reachable when a SINGLE
|
|
800
|
+
* message serializes above the budget (an oversized paste): the alternative is
|
|
801
|
+
* a provider rejection that no retry can clear, which strands the session with
|
|
802
|
+
* a full window forever.
|
|
803
|
+
*/
|
|
804
|
+
function clampConversationToBudget(text: string, budgetTokens: number, tokens: number): string {
|
|
805
|
+
if (tokens <= budgetTokens) return text;
|
|
806
|
+
const keep = Math.max(1024, Math.floor((text.length * budgetTokens * 0.95) / tokens));
|
|
807
|
+
if (keep >= text.length) return text;
|
|
808
|
+
return `${text.slice(0, keep)}\n\n[... ${text.length - keep} more characters truncated]`;
|
|
809
|
+
}
|
|
810
|
+
|
|
811
|
+
/** One planned summarization call: its messages and the budget they were packed for. */
|
|
812
|
+
interface SummaryWindow {
|
|
813
|
+
messages: Message[];
|
|
814
|
+
budgetTokens: number;
|
|
815
|
+
/** Serialization reused from the fit check, so the common path serializes once. */
|
|
816
|
+
text?: string;
|
|
817
|
+
}
|
|
818
|
+
|
|
819
|
+
/**
|
|
820
|
+
* Partition a conversation into windows that each fit `budgetTokens`, splitting
|
|
821
|
+
* on message boundaries. Only called when the whole conversation does not fit —
|
|
822
|
+
* the common single-window path never pays this per-message sizing pass.
|
|
823
|
+
*/
|
|
824
|
+
function planSummaryWindows(
|
|
825
|
+
messages: Message[],
|
|
826
|
+
tokenizer: Tokenizer,
|
|
827
|
+
dialect: Dialect | undefined,
|
|
828
|
+
budgetTokens: number,
|
|
829
|
+
): Message[][] {
|
|
830
|
+
const windows: Message[][] = [];
|
|
831
|
+
let current: Message[] = [];
|
|
832
|
+
let currentTokens = 0;
|
|
833
|
+
for (const message of messages) {
|
|
834
|
+
const tokens = tokenizer.countTokens(serializeConversationForSummary([message], dialect));
|
|
835
|
+
if (currentTokens > 0 && currentTokens + tokens > budgetTokens) {
|
|
836
|
+
windows.push(current);
|
|
837
|
+
current = [];
|
|
838
|
+
currentTokens = 0;
|
|
839
|
+
}
|
|
840
|
+
current.push(message);
|
|
841
|
+
currentTokens += tokens;
|
|
842
|
+
}
|
|
843
|
+
if (current.length > 0) windows.push(current);
|
|
844
|
+
return windows;
|
|
845
|
+
}
|
|
846
|
+
|
|
890
847
|
export async function generateSummary(
|
|
891
848
|
currentMessages: AgentMessage[],
|
|
892
849
|
model: Model,
|
|
@@ -899,6 +856,88 @@ export async function generateSummary(
|
|
|
899
856
|
): Promise<string> {
|
|
900
857
|
const maxTokens = Math.min(Math.floor(0.8 * reserveTokens), MAX_SUMMARY_TOKENS);
|
|
901
858
|
|
|
859
|
+
// Serialize conversation to text so model doesn't try to continue it
|
|
860
|
+
// Convert to LLM messages first (handles custom app messages when caller provides a transformer).
|
|
861
|
+
const llmMessages = (options?.convertToLlm ?? defaultConvertToLlm)(currentMessages);
|
|
862
|
+
const dialect = preferredDialect(model.id);
|
|
863
|
+
const tokenizer = new Tokenizer(model);
|
|
864
|
+
const wholeConversation = serializeConversationForSummary(llmMessages, dialect);
|
|
865
|
+
const budgetTokens = summaryInputBudgetTokens(model, maxTokens);
|
|
866
|
+
// A span that outgrew the summarizer's window is summarized as a fold: each
|
|
867
|
+
// window updates the summary carried out of the previous one, which is the
|
|
868
|
+
// same contract the update prompt already implements for iterative
|
|
869
|
+
// compaction. The alternative is a hard provider rejection on a prompt no
|
|
870
|
+
// retry can shrink — the state a cross-provider compaction boundary
|
|
871
|
+
// (see `prepareCompaction`) puts a long session into. One window is the
|
|
872
|
+
// common case and costs exactly the one call it always did.
|
|
873
|
+
const pending: SummaryWindow[] = tokenizer.checkTokenBudget(wholeConversation, budgetTokens).fits
|
|
874
|
+
? [{ messages: llmMessages, budgetTokens, text: wholeConversation }]
|
|
875
|
+
: planSummaryWindows(llmMessages, tokenizer, dialect, budgetTokens).map(messages => ({ messages, budgetTokens }));
|
|
876
|
+
|
|
877
|
+
let carriedSummary = previousSummary;
|
|
878
|
+
while (pending.length > 0) {
|
|
879
|
+
const window = pending[0];
|
|
880
|
+
const text = window.text ?? serializeConversationForSummary(window.messages, dialect);
|
|
881
|
+
// A budget probe, not a raw count: a window whose bytes already fit needs
|
|
882
|
+
// neither an exact count nor the clamp, and the bust path hands back the
|
|
883
|
+
// exact count the proportional clamp needs as its denominator.
|
|
884
|
+
const budget = tokenizer.checkTokenBudget(text, window.budgetTokens);
|
|
885
|
+
try {
|
|
886
|
+
carriedSummary = await summarizeConversationWindow(
|
|
887
|
+
budget.fits ? text : clampConversationToBudget(text, window.budgetTokens, budget.tokens),
|
|
888
|
+
carriedSummary,
|
|
889
|
+
model,
|
|
890
|
+
maxTokens,
|
|
891
|
+
apiKey,
|
|
892
|
+
signal,
|
|
893
|
+
customInstructions,
|
|
894
|
+
options,
|
|
895
|
+
);
|
|
896
|
+
} catch (error) {
|
|
897
|
+
// The catalog window can overstate what the provider actually accepts:
|
|
898
|
+
// `claude-sonnet-4-5` advertises 1M but is beta-gated to 200k on OAuth
|
|
899
|
+
// credentials (see `anthropic.ts` — the 1M beta is never advertised).
|
|
900
|
+
// Halve and re-plan rather than failing the whole compaction on a
|
|
901
|
+
// window size only the provider can tell us is wrong.
|
|
902
|
+
// Halve what was actually SENT, not the budget it was planned against:
|
|
903
|
+
// the rejection proves the plan was fiction, so converging on the real
|
|
904
|
+
// cap must not spend a call per level of an imaginary ladder. The cheap
|
|
905
|
+
// fit path never counted this window, so pay for the exact size here —
|
|
906
|
+
// one tokenization is nothing against the provider round trip already lost.
|
|
907
|
+
const sentTokens = budget.exact ? budget.tokens : tokenizer.countTokens(text, "strict");
|
|
908
|
+
const halved = Math.floor(Math.min(window.budgetTokens, sentTokens) / 2);
|
|
909
|
+
if (
|
|
910
|
+
!AIError.is(AIError.classify(error), AIError.Flag.ContextOverflow) ||
|
|
911
|
+
halved < minSummaryInputTokens(model)
|
|
912
|
+
) {
|
|
913
|
+
throw error;
|
|
914
|
+
}
|
|
915
|
+
pending.splice(
|
|
916
|
+
0,
|
|
917
|
+
1,
|
|
918
|
+
...planSummaryWindows(window.messages, tokenizer, dialect, halved).map(messages => ({
|
|
919
|
+
messages,
|
|
920
|
+
budgetTokens: halved,
|
|
921
|
+
})),
|
|
922
|
+
);
|
|
923
|
+
continue;
|
|
924
|
+
}
|
|
925
|
+
pending.shift();
|
|
926
|
+
}
|
|
927
|
+
return carriedSummary ?? "";
|
|
928
|
+
}
|
|
929
|
+
|
|
930
|
+
/** One summarization call over a single conversation window. */
|
|
931
|
+
async function summarizeConversationWindow(
|
|
932
|
+
conversationText: string,
|
|
933
|
+
previousSummary: string | undefined,
|
|
934
|
+
model: Model,
|
|
935
|
+
maxTokens: number,
|
|
936
|
+
apiKey: ApiKey,
|
|
937
|
+
signal: AbortSignal | undefined,
|
|
938
|
+
customInstructions: string | undefined,
|
|
939
|
+
options: SummaryOptions | undefined,
|
|
940
|
+
): Promise<string> {
|
|
902
941
|
// Use update prompt if we have a previous summary, otherwise initial prompt
|
|
903
942
|
let basePrompt = previousSummary ? UPDATE_SUMMARIZATION_PROMPT : SUMMARIZATION_PROMPT;
|
|
904
943
|
if (options?.promptOverride) {
|
|
@@ -908,15 +947,10 @@ export async function generateSummary(
|
|
|
908
947
|
basePrompt = `${basePrompt}\n\nAdditional focus: ${customInstructions}`;
|
|
909
948
|
}
|
|
910
949
|
|
|
911
|
-
// Serialize conversation to text so model doesn't try to continue it
|
|
912
|
-
// Convert to LLM messages first (handles custom app messages when caller provides a transformer).
|
|
913
|
-
const llmMessages = (options?.convertToLlm ?? defaultConvertToLlm)(currentMessages);
|
|
914
|
-
const conversationText = serializeConversationForSummary(llmMessages, preferredDialect(model.id));
|
|
915
|
-
|
|
916
950
|
// Build the prompt with conversation wrapped in tags
|
|
917
951
|
let promptText = `<conversation>\n${conversationText}\n</conversation>\n\n`;
|
|
918
952
|
if (previousSummary) {
|
|
919
|
-
promptText += `<previous-summary>\n${previousSummary}\n</previous-summary>\n\n`;
|
|
953
|
+
promptText += `<previous-summary>\n${escapeSummaryBoundaryTags(previousSummary)}\n</previous-summary>\n\n`;
|
|
920
954
|
}
|
|
921
955
|
promptText += formatAdditionalContext(options?.extraContext);
|
|
922
956
|
promptText += basePrompt;
|
|
@@ -1135,7 +1169,7 @@ async function generateShortSummary(
|
|
|
1135
1169
|
|
|
1136
1170
|
let promptText = `<conversation>\n${conversationText}\n</conversation>\n\n`;
|
|
1137
1171
|
if (historySummary) {
|
|
1138
|
-
promptText += `<previous-summary>\n${historySummary}\n</previous-summary>\n\n`;
|
|
1172
|
+
promptText += `<previous-summary>\n${escapeSummaryBoundaryTags(historySummary)}\n</previous-summary>\n\n`;
|
|
1139
1173
|
}
|
|
1140
1174
|
promptText += formatAdditionalContext(options?.extraContext);
|
|
1141
1175
|
promptText += SHORT_SUMMARY_PROMPT;
|
|
@@ -1234,7 +1268,7 @@ export interface CompactionPreparation {
|
|
|
1234
1268
|
* let the active model replay it, so keying reuse on "any candidate shares the
|
|
1235
1269
|
* provider" left a provider-switched session permanently context-less (#6343).
|
|
1236
1270
|
*/
|
|
1237
|
-
function remotePreserveReusable(
|
|
1271
|
+
export function remotePreserveReusable(
|
|
1238
1272
|
preserveData: Record<string, unknown> | undefined,
|
|
1239
1273
|
activeModel: Model,
|
|
1240
1274
|
settings: CompactionSettings,
|
|
@@ -1247,38 +1281,75 @@ function remotePreserveReusable(
|
|
|
1247
1281
|
return v2Ok || shouldUseOpenAiRemoteCompaction(activeModel);
|
|
1248
1282
|
}
|
|
1249
1283
|
|
|
1284
|
+
/**
|
|
1285
|
+
* Index of the newest compaction entry the active model can actually read, or
|
|
1286
|
+
* `-1` when none can.
|
|
1287
|
+
*
|
|
1288
|
+
* A provider-native remote compaction (V2 or V1) stores an opaque replay payload
|
|
1289
|
+
* and only a placeholder summary, so for any OTHER provider that entry
|
|
1290
|
+
* summarizes nothing and the history behind it is still live context. Callers
|
|
1291
|
+
* must therefore treat it as absent: `prepareCompaction` re-expands past it and
|
|
1292
|
+
* summarizes those messages locally, and the maintenance ops that use the
|
|
1293
|
+
* compaction boundary to skip "already summarized away" entries must not skip
|
|
1294
|
+
* entries that no summary covers.
|
|
1295
|
+
*/
|
|
1296
|
+
export function findReadableCompactionIndex(
|
|
1297
|
+
pathEntries: SessionEntry[],
|
|
1298
|
+
settings: CompactionSettings,
|
|
1299
|
+
activeModel?: Model,
|
|
1300
|
+
): number {
|
|
1301
|
+
for (let i = pathEntries.length - 1; i >= 0; i--) {
|
|
1302
|
+
if (pathEntries[i].type !== "compaction") continue;
|
|
1303
|
+
const entry = pathEntries[i] as CompactionEntry;
|
|
1304
|
+
if (activeModel && !remotePreserveReusable(entry.preserveData, activeModel, settings)) continue;
|
|
1305
|
+
return i;
|
|
1306
|
+
}
|
|
1307
|
+
return -1;
|
|
1308
|
+
}
|
|
1309
|
+
|
|
1310
|
+
/**
|
|
1311
|
+
* Pass the caller's warm `tokenizer` (the Agent's for the active model) so the
|
|
1312
|
+
* full-branch estimate walk hits its memo; the cold default is for one-shot
|
|
1313
|
+
* callers that have no live agent.
|
|
1314
|
+
*/
|
|
1250
1315
|
export function prepareCompaction(
|
|
1251
1316
|
pathEntries: SessionEntry[],
|
|
1252
1317
|
settings: CompactionSettings,
|
|
1253
1318
|
activeModel?: Model,
|
|
1319
|
+
tokenizer: Tokenizer = new Tokenizer(activeModel),
|
|
1254
1320
|
): CompactionPreparation | undefined {
|
|
1255
1321
|
if (pathEntries.length > 0 && pathEntries[pathEntries.length - 1].type === "compaction") {
|
|
1256
1322
|
return undefined;
|
|
1257
1323
|
}
|
|
1258
1324
|
|
|
1259
|
-
let prevCompactionIndex =
|
|
1260
|
-
|
|
1261
|
-
|
|
1262
|
-
|
|
1263
|
-
|
|
1264
|
-
|
|
1265
|
-
|
|
1266
|
-
|
|
1267
|
-
|
|
1268
|
-
|
|
1269
|
-
|
|
1325
|
+
let prevCompactionIndex = findReadableCompactionIndex(pathEntries, settings, activeModel);
|
|
1326
|
+
|
|
1327
|
+
// Honor the latest `/clear` reset boundary. `/clear` records a
|
|
1328
|
+
// `reset_boundary` marker and reports the model context empty, so compaction
|
|
1329
|
+
// must not resurrect the dropped pre-clear turns into its summary — matching
|
|
1330
|
+
// how buildSessionContext starts the model-context rebuild after the boundary.
|
|
1331
|
+
// A boundary after the last reusable compaction supersedes it: the pre-reset
|
|
1332
|
+
// summary was cleared too, so drop the previous-compaction reuse and start
|
|
1333
|
+
// fresh after the boundary. A boundary at or before that compaction is already
|
|
1334
|
+
// superseded by it, so only scan newer entries.
|
|
1335
|
+
let resetBoundaryIndex = -1;
|
|
1336
|
+
for (let i = pathEntries.length - 1; i > prevCompactionIndex; i--) {
|
|
1337
|
+
if (pathEntries[i].type === "reset_boundary") {
|
|
1338
|
+
resetBoundaryIndex = i;
|
|
1339
|
+
break;
|
|
1270
1340
|
}
|
|
1271
|
-
prevCompactionIndex = i;
|
|
1272
|
-
break;
|
|
1273
1341
|
}
|
|
1274
|
-
|
|
1342
|
+
if (resetBoundaryIndex > prevCompactionIndex) {
|
|
1343
|
+
prevCompactionIndex = -1;
|
|
1344
|
+
}
|
|
1345
|
+
const boundaryStart = Math.max(prevCompactionIndex, resetBoundaryIndex) + 1;
|
|
1275
1346
|
const boundaryEnd = pathEntries.length;
|
|
1276
1347
|
|
|
1277
1348
|
const lastUsage = getLastAssistantUsage(pathEntries);
|
|
1278
1349
|
const tokensBefore = lastUsage ? calculateContextTokens(lastUsage) : 0;
|
|
1279
1350
|
let keepRecentTokens = settings.keepRecentTokens;
|
|
1280
1351
|
if (lastUsage) {
|
|
1281
|
-
const estimatedTokens = estimateEntriesTokens(pathEntries, boundaryStart, boundaryEnd);
|
|
1352
|
+
const estimatedTokens = estimateEntriesTokens(pathEntries, tokenizer, boundaryStart, boundaryEnd);
|
|
1282
1353
|
const promptTokens = calculatePromptTokens(lastUsage);
|
|
1283
1354
|
const ratio = estimatedTokens > 0 ? promptTokens / estimatedTokens : 0;
|
|
1284
1355
|
if (Number.isFinite(ratio) && ratio > 1) {
|
|
@@ -1286,7 +1357,7 @@ export function prepareCompaction(
|
|
|
1286
1357
|
}
|
|
1287
1358
|
}
|
|
1288
1359
|
|
|
1289
|
-
const cutPoint = findCutPoint(pathEntries, boundaryStart, boundaryEnd, keepRecentTokens);
|
|
1360
|
+
const cutPoint = findCutPoint(pathEntries, tokenizer, boundaryStart, boundaryEnd, keepRecentTokens);
|
|
1290
1361
|
|
|
1291
1362
|
// Get ID of first kept entry
|
|
1292
1363
|
const firstKeptEntry = pathEntries[cutPoint.firstKeptEntryIndex];
|
|
@@ -1525,6 +1596,7 @@ export async function compact(
|
|
|
1525
1596
|
: undefined;
|
|
1526
1597
|
const trimmed = trimRemoteCompactionInputToContextWindow(
|
|
1527
1598
|
remoteHistory,
|
|
1599
|
+
new Tokenizer(model),
|
|
1528
1600
|
model.contextWindow,
|
|
1529
1601
|
instructions,
|
|
1530
1602
|
tools,
|
|
@@ -117,6 +117,16 @@ export interface ModeChangeEntry extends SessionEntryBase {
|
|
|
117
117
|
data?: Record<string, unknown>;
|
|
118
118
|
}
|
|
119
119
|
|
|
120
|
+
/**
|
|
121
|
+
* Durable context-reset marker recorded by an in-place `/clear`. It carries no
|
|
122
|
+
* payload — its presence on the branch means every entry before it was dropped
|
|
123
|
+
* from the model context, so context assembly and compaction start after the
|
|
124
|
+
* latest one. The full pre-reset history stays on disk for transcript export.
|
|
125
|
+
*/
|
|
126
|
+
export interface ResetBoundaryEntry extends SessionEntryBase {
|
|
127
|
+
type: "reset_boundary";
|
|
128
|
+
}
|
|
129
|
+
|
|
120
130
|
export interface CustomCompactionSessionEntries {}
|
|
121
131
|
|
|
122
132
|
export type SessionEntry =
|
|
@@ -133,6 +143,7 @@ export type SessionEntry =
|
|
|
133
143
|
| TtsrInjectionEntry
|
|
134
144
|
| SessionInitEntry
|
|
135
145
|
| ModeChangeEntry
|
|
146
|
+
| ResetBoundaryEntry
|
|
136
147
|
| CustomCompactionSessionEntries[keyof CustomCompactionSessionEntries];
|
|
137
148
|
|
|
138
149
|
export interface ReadonlySessionManager {
|