@oh-my-pi/pi-agent-core 17.3.7 → 17.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -9,8 +9,8 @@ import type { Api, ApiKey, AssistantMessage, Context, Model, SimpleStreamOptions
9
9
  import { preferredDialect } from "@oh-my-pi/pi-catalog/identity";
10
10
  import { prompt } from "@oh-my-pi/pi-utils";
11
11
  import { type AgentTelemetry, instrumentedCompleteSimple } from "../telemetry";
12
+ import { Tokenizer } from "../tokenizer";
12
13
  import type { AgentMessage } from "../types";
13
- import { estimateTokens } from "./compaction";
14
14
  import type { ReadonlySessionManager, SessionEntry } from "./entries";
15
15
  import {
16
16
  type ConvertToLlm,
@@ -191,7 +191,9 @@ function getMessageFromEntry(entry: SessionEntry): AgentMessage | undefined {
191
191
  return createBranchSummaryMessage(entry.summary, entry.fromId, entry.timestamp);
192
192
 
193
193
  case "compaction":
194
- return createCompactionSummaryMessage(entry.summary, entry.tokensBefore, entry.timestamp, entry.shortSummary);
194
+ return createCompactionSummaryMessage(entry.summary, entry.tokensBefore, entry.timestamp, {
195
+ shortSummary: entry.shortSummary,
196
+ });
195
197
 
196
198
  // These don't contribute to conversation content
197
199
  case "thinking_level_change":
@@ -206,14 +208,14 @@ function getMessageFromEntry(entry: SessionEntry): AgentMessage | undefined {
206
208
  }
207
209
  }
208
210
 
209
- function estimateBranchSummaryTokens(message: AgentMessage): number {
210
- if (message.role !== "toolResult") return estimateTokens(message);
211
+ function estimateBranchSummaryTokens(message: AgentMessage, tokenizer: Tokenizer): number {
212
+ if (message.role !== "toolResult") return tokenizer.countMessage(message);
211
213
  const text = message.content
212
214
  .filter((c): c is { type: "text"; text: string } => c.type === "text")
213
215
  .map(c => c.text)
214
216
  .join("");
215
217
  if (!text) return 0;
216
- return estimateTokens({
218
+ return tokenizer.countMessage({
217
219
  ...message,
218
220
  content: [{ type: "text", text: truncateToolResultForSummary(text) }],
219
221
  });
@@ -232,7 +234,11 @@ function estimateBranchSummaryTokens(message: AgentMessage): number {
232
234
  * @param entries - Entries in chronological order
233
235
  * @param tokenBudget - Maximum tokens to include (0 = no limit)
234
236
  */
235
- export function prepareBranchEntries(entries: SessionEntry[], tokenBudget: number = 0): BranchPreparation {
237
+ export function prepareBranchEntries(
238
+ entries: SessionEntry[],
239
+ tokenizer: Tokenizer,
240
+ tokenBudget: number = 0,
241
+ ): BranchPreparation {
236
242
  const messages: AgentMessage[] = [];
237
243
  const fileOps = createFileOps();
238
244
  let totalTokens = 0;
@@ -264,7 +270,7 @@ export function prepareBranchEntries(entries: SessionEntry[], tokenBudget: numbe
264
270
  // Extract file ops from assistant messages (tool calls)
265
271
  extractFileOpsFromMessage(message, fileOps);
266
272
 
267
- const tokens = estimateBranchSummaryTokens(message);
273
+ const tokens = estimateBranchSummaryTokens(message, tokenizer);
268
274
 
269
275
  // Check budget before adding
270
276
  if (tokenBudget > 0 && totalTokens + tokens > tokenBudget) {
@@ -309,8 +315,9 @@ export async function generateBranchSummary(
309
315
  // Token budget = context window minus reserved space for prompt + response
310
316
  const contextWindow = model.contextWindow || 128000;
311
317
  const tokenBudget = contextWindow - reserveTokens;
318
+ const tokenizer = new Tokenizer(model);
312
319
 
313
- const { messages, fileOps } = prepareBranchEntries(entries, tokenBudget);
320
+ const { messages, fileOps } = prepareBranchEntries(entries, tokenizer, tokenBudget);
314
321
 
315
322
  if (messages.length === 0) {
316
323
  return { summary: "No content to summarize" };
@@ -23,6 +23,7 @@ import {
23
23
  type Usage,
24
24
  withAuth,
25
25
  } from "@oh-my-pi/pi-ai";
26
+ import type { Dialect } from "@oh-my-pi/pi-ai/dialect";
26
27
  import * as AIError from "@oh-my-pi/pi-ai/error";
27
28
  import { createOpenAICodexCompactionRequestContext } from "@oh-my-pi/pi-ai/providers/openai-codex-responses";
28
29
  import { convertTools } from "@oh-my-pi/pi-ai/providers/openai-responses";
@@ -30,11 +31,11 @@ import { buildResponsesInput, resolveOpenAICompatPolicy } from "@oh-my-pi/pi-ai/
30
31
  import { stripOpenAIResponsesOutputOnlyStatusesForReplay } from "@oh-my-pi/pi-ai/utils";
31
32
  import { preferredDialect } from "@oh-my-pi/pi-catalog/identity";
32
33
  import { clampThinkingLevelForModel } from "@oh-my-pi/pi-catalog/model-thinking";
33
- import { isRecord, logger, prompt, stringifyJson } from "@oh-my-pi/pi-utils";
34
+ import { isRecord, logger, prompt } from "@oh-my-pi/pi-utils";
34
35
  import * as snapcompact from "@oh-my-pi/snapcompact";
35
36
  import { type AgentTelemetry, instrumentedCompleteSimple } from "../telemetry";
36
37
  import { ThinkingLevel } from "../thinking";
37
- import { countTokens } from "../tokenizer";
38
+ import { Tokenizer } from "../tokenizer";
38
39
  import type { AgentMessage } from "../types";
39
40
  import {
40
41
  buildCompactionV2Request,
@@ -46,7 +47,6 @@ import {
46
47
  } from "./compaction-v2-streaming";
47
48
  import type { CompactionEntry, SessionEntry } from "./entries";
48
49
  import { NativeCompactionError } from "./errors";
49
- import { isEstimateCacheable, readEstimateCache, writeEstimateCache } from "./message-cache";
50
50
  import { type ConvertToLlm, createBranchSummaryMessage, createCustomMessage, defaultConvertToLlm } from "./messages";
51
51
  import {
52
52
  buildOpenAiNativeHistory,
@@ -68,6 +68,7 @@ import snapcompactArchiveContextPrompt from "./prompts/snapcompact-archive-conte
68
68
  import {
69
69
  computeFileLists,
70
70
  createFileOps,
71
+ escapeSummaryBoundaryTags,
71
72
  extractFileOpsFromMessage,
72
73
  type FileOperations,
73
74
  SUMMARIZATION_SYSTEM_PROMPT,
@@ -388,144 +389,17 @@ export function resolveThresholdTokens(contextWindow: number, settings: Compacti
388
389
  // Cut point detection
389
390
  // ============================================================================
390
391
 
391
- /**
392
- * Image content has no tokenizer representation; charge a fixed estimate
393
- * matching what providers typically bill for inline images.
394
- */
395
- const IMAGE_TOKEN_ESTIMATE = 1200;
396
-
397
- /**
398
- * Estimate token count for a message using cl100k_base via the native
399
- * tokenizer. This is not Claude's first-party tokenizer (Anthropic doesn't
400
- * publish one) but is within ~5–10% across English/code text.
401
- *
402
- * `excludeEncryptedReasoning` drops opaque provider reasoning payloads
403
- * (`thinkingSignature`, `redactedThinking`) from the estimate. Those are billed
404
- * by the provider on replay, so the default counts them — but their *local*
405
- * byte size can diverge wildly from what the provider charges, so the
406
- * compaction floor (which only needs the reliably-countable, on-wire-compressible
407
- * content) excludes them to avoid false triggers on thinking-heavy turns.
408
- */
409
- export function estimateTokens(message: AgentMessage, options?: { excludeEncryptedReasoning?: boolean }): number {
410
- // Settled historical messages are counted once and reused until an owner
411
- // (prune/shake/strip-images) invalidates them; streaming assistants bypass
412
- // the cache entirely (see message-cache.ts settle-gate invariant).
413
- const cacheable = isEstimateCacheable(message);
414
- const excludeEncryptedReasoning = options?.excludeEncryptedReasoning === true;
415
- if (cacheable) {
416
- const cached = readEstimateCache(message, excludeEncryptedReasoning);
417
- if (cached !== undefined) return cached;
418
- }
419
- const result = computeMessageTokens(message, options);
420
- if (cacheable) writeEstimateCache(message, excludeEncryptedReasoning, result);
421
- return result;
422
- }
423
-
424
- function computeMessageTokens(message: AgentMessage, options?: { excludeEncryptedReasoning?: boolean }): number {
425
- const fragments: string[] = [];
426
- let extra = 0;
427
- if ((message as { role?: string }).role === "bashExecution") {
428
- const bash = message as { command?: unknown; output?: unknown };
429
- if (typeof bash.command === "string") fragments.push(bash.command);
430
- if (typeof bash.output === "string") fragments.push(bash.output);
431
- return fragments.length === 0 ? 0 : countTokens(fragments);
432
- }
433
-
434
- switch (message.role) {
435
- case "user": {
436
- const content = (message as { content: string | Array<{ type: string; text?: string }> }).content;
437
- if (typeof content === "string") {
438
- fragments.push(content);
439
- } else if (Array.isArray(content)) {
440
- for (const block of content) {
441
- if (block.type === "text" && block.text) {
442
- fragments.push(block.text);
443
- }
444
- }
445
- }
446
- break;
447
- }
448
- case "assistant": {
449
- const assistant = message as AssistantMessage;
450
- for (const block of assistant.content) {
451
- if (block.type === "text") {
452
- fragments.push(block.text);
453
- } else if (block.type === "thinking") {
454
- fragments.push(block.thinking);
455
- // Providers charge for the opaque signature/reasoning payload that
456
- // rides alongside the thinking text (OpenAI Responses encrypted
457
- // reasoning items, Anthropic signed thinking blocks, etc.). Without
458
- // counting it, this estimator can read ~half of the provider-reported
459
- // usage on thinking-heavy turns — see #2275 for the resulting
460
- // compaction-trigger / post-check metric divergence. The compaction
461
- // floor excludes it (its local byte size diverges from provider billing).
462
- if (block.thinkingSignature && !options?.excludeEncryptedReasoning) {
463
- fragments.push(block.thinkingSignature);
464
- }
465
- } else if (block.type === "toolCall") {
466
- fragments.push(block.name);
467
- fragments.push(stringifyJson(block.arguments) ?? "null");
468
- } else if (block.type === "redactedThinking") {
469
- // Encrypted reasoning blob the provider still bills for on replay;
470
- // excluded from the compaction floor for the same reason as above.
471
- if (!options?.excludeEncryptedReasoning) fragments.push(block.data);
472
- } else if (block.type === "anthropicServerTool") {
473
- // Native Anthropic server-tool call/result replayed verbatim on the
474
- // wire (server_tool_use input and opaque result content). This opaque
475
- // provider-replay state the provider still
476
- // bills for on same-provider replay; excluded from the compaction
477
- // floor like other encrypted reasoning because its local byte size
478
- // diverges from provider billing.
479
- if (!options?.excludeEncryptedReasoning) fragments.push(stringifyJson(block.block) ?? "null");
480
- }
481
- }
482
- break;
483
- }
484
- case "hookMessage":
485
- case "toolResult": {
486
- if (typeof message.content === "string") {
487
- fragments.push(message.content);
488
- } else {
489
- for (const block of message.content) {
490
- if (block.type === "text" && block.text) {
491
- fragments.push(block.text);
492
- } else if (block.type === "image") {
493
- extra += IMAGE_TOKEN_ESTIMATE;
494
- }
495
- }
496
- }
497
- break;
498
- }
499
- case "branchSummary":
500
- case "compactionSummary": {
501
- fragments.push(message.summary);
502
- if (message.role === "compactionSummary") {
503
- if (message.blocks) {
504
- for (const block of message.blocks) {
505
- if (block.type === "text") fragments.push(block.text);
506
- else extra += snapcompact.FRAME_TOKEN_ESTIMATE;
507
- }
508
- } else if (message.images) {
509
- // Snapcompact frames render at ≥1568px; providers bill the downscaled cap.
510
- extra += message.images.length * snapcompact.FRAME_TOKEN_ESTIMATE;
511
- }
512
- }
513
- break;
514
- }
515
- default:
516
- return 0;
517
- }
518
-
519
- if (fragments.length === 0) return extra;
520
- return extra + countTokens(fragments);
521
- }
522
-
523
- function estimateEntriesTokens(entries: SessionEntry[], startIndex: number, endIndex: number): number {
392
+ function estimateEntriesTokens(
393
+ entries: SessionEntry[],
394
+ tokenizer: Tokenizer,
395
+ startIndex: number,
396
+ endIndex: number,
397
+ ): number {
524
398
  let total = 0;
525
399
  for (let i = startIndex; i < endIndex; i++) {
526
400
  const msg = getMessageFromEntry(entries[i]);
527
401
  if (msg) {
528
- total += estimateTokens(msg);
402
+ total += tokenizer.countMessage(msg);
529
403
  }
530
404
  }
531
405
  return total;
@@ -624,6 +498,7 @@ export interface CutPointResult {
624
498
  */
625
499
  export function findCutPoint(
626
500
  entries: SessionEntry[],
501
+ tokenizer: Tokenizer,
627
502
  startIndex: number,
628
503
  endIndex: number,
629
504
  keepRecentTokens: number,
@@ -643,7 +518,7 @@ export function findCutPoint(
643
518
  if (entry.type !== "message") continue;
644
519
 
645
520
  // Estimate this message's size
646
- const messageTokens = estimateTokens(entry.message);
521
+ const messageTokens = tokenizer.countMessage(entry.message);
647
522
  accumulatedTokens += messageTokens;
648
523
 
649
524
  // Check if we've exceeded the budget
@@ -887,6 +762,88 @@ function createSnapcompactArchiveMigrationMessage(archiveText: string): Message
887
762
  };
888
763
  }
889
764
 
765
+ /**
766
+ * Fallback window for a model whose catalog entry carries no usable context
767
+ * window; matches the smallest window any compaction-capable model ships with.
768
+ */
769
+ const DEFAULT_SUMMARY_INPUT_WINDOW = 200_000;
770
+
771
+ /**
772
+ * Floor for one summarization window, so a tiny model still makes progress.
773
+ * Scaled down (never below 1k) for models whose window cannot host the full
774
+ * floor next to the carried summary and output reserves.
775
+ */
776
+ const MIN_SUMMARY_INPUT_TOKENS = 16_384;
777
+
778
+ /** Smallest window worth planning for `model`; below this, overflow recovery gives up. */
779
+ function minSummaryInputTokens(model: Model): number {
780
+ const window = model.contextWindow && model.contextWindow > 0 ? model.contextWindow : DEFAULT_SUMMARY_INPUT_WINDOW;
781
+ return Math.min(MIN_SUMMARY_INPUT_TOKENS, Math.max(1_024, Math.floor(window / 8)));
782
+ }
783
+
784
+ /**
785
+ * Usable conversation input for ONE summarization call: the summarizer's window
786
+ * minus the summary it must emit, the previous summary it carries forward, and
787
+ * prompt scaffolding. Providers tokenize differently from the local cl100k
788
+ * estimate, so the window is discounted before the fixed reserves come off.
789
+ */
790
+ function summaryInputBudgetTokens(model: Model, maxTokens: number): number {
791
+ const window = model.contextWindow && model.contextWindow > 0 ? model.contextWindow : DEFAULT_SUMMARY_INPUT_WINDOW;
792
+ // 0.8, not "window minus reserves": provider tokenizers disagree with the
793
+ // local cl100k estimate by a few percent, and being wrong here is a hard
794
+ // 400 on the one call that is supposed to rescue an oversized session.
795
+ return Math.max(minSummaryInputTokens(model), Math.floor(window * 0.8) - maxTokens - MAX_SUMMARY_TOKENS);
796
+ }
797
+
798
+ /**
799
+ * Clamp one serialized window to the budget. Only reachable when a SINGLE
800
+ * message serializes above the budget (an oversized paste): the alternative is
801
+ * a provider rejection that no retry can clear, which strands the session with
802
+ * a full window forever.
803
+ */
804
+ function clampConversationToBudget(text: string, budgetTokens: number, tokens: number): string {
805
+ if (tokens <= budgetTokens) return text;
806
+ const keep = Math.max(1024, Math.floor((text.length * budgetTokens * 0.95) / tokens));
807
+ if (keep >= text.length) return text;
808
+ return `${text.slice(0, keep)}\n\n[... ${text.length - keep} more characters truncated]`;
809
+ }
810
+
811
+ /** One planned summarization call: its messages and the budget they were packed for. */
812
+ interface SummaryWindow {
813
+ messages: Message[];
814
+ budgetTokens: number;
815
+ /** Serialization reused from the fit check, so the common path serializes once. */
816
+ text?: string;
817
+ }
818
+
819
+ /**
820
+ * Partition a conversation into windows that each fit `budgetTokens`, splitting
821
+ * on message boundaries. Only called when the whole conversation does not fit —
822
+ * the common single-window path never pays this per-message sizing pass.
823
+ */
824
+ function planSummaryWindows(
825
+ messages: Message[],
826
+ tokenizer: Tokenizer,
827
+ dialect: Dialect | undefined,
828
+ budgetTokens: number,
829
+ ): Message[][] {
830
+ const windows: Message[][] = [];
831
+ let current: Message[] = [];
832
+ let currentTokens = 0;
833
+ for (const message of messages) {
834
+ const tokens = tokenizer.countTokens(serializeConversationForSummary([message], dialect));
835
+ if (currentTokens > 0 && currentTokens + tokens > budgetTokens) {
836
+ windows.push(current);
837
+ current = [];
838
+ currentTokens = 0;
839
+ }
840
+ current.push(message);
841
+ currentTokens += tokens;
842
+ }
843
+ if (current.length > 0) windows.push(current);
844
+ return windows;
845
+ }
846
+
890
847
  export async function generateSummary(
891
848
  currentMessages: AgentMessage[],
892
849
  model: Model,
@@ -899,6 +856,88 @@ export async function generateSummary(
899
856
  ): Promise<string> {
900
857
  const maxTokens = Math.min(Math.floor(0.8 * reserveTokens), MAX_SUMMARY_TOKENS);
901
858
 
859
+ // Serialize conversation to text so model doesn't try to continue it
860
+ // Convert to LLM messages first (handles custom app messages when caller provides a transformer).
861
+ const llmMessages = (options?.convertToLlm ?? defaultConvertToLlm)(currentMessages);
862
+ const dialect = preferredDialect(model.id);
863
+ const tokenizer = new Tokenizer(model);
864
+ const wholeConversation = serializeConversationForSummary(llmMessages, dialect);
865
+ const budgetTokens = summaryInputBudgetTokens(model, maxTokens);
866
+ // A span that outgrew the summarizer's window is summarized as a fold: each
867
+ // window updates the summary carried out of the previous one, which is the
868
+ // same contract the update prompt already implements for iterative
869
+ // compaction. The alternative is a hard provider rejection on a prompt no
870
+ // retry can shrink — the state a cross-provider compaction boundary
871
+ // (see `prepareCompaction`) puts a long session into. One window is the
872
+ // common case and costs exactly the one call it always did.
873
+ const pending: SummaryWindow[] = tokenizer.checkTokenBudget(wholeConversation, budgetTokens).fits
874
+ ? [{ messages: llmMessages, budgetTokens, text: wholeConversation }]
875
+ : planSummaryWindows(llmMessages, tokenizer, dialect, budgetTokens).map(messages => ({ messages, budgetTokens }));
876
+
877
+ let carriedSummary = previousSummary;
878
+ while (pending.length > 0) {
879
+ const window = pending[0];
880
+ const text = window.text ?? serializeConversationForSummary(window.messages, dialect);
881
+ // A budget probe, not a raw count: a window whose bytes already fit needs
882
+ // neither an exact count nor the clamp, and the bust path hands back the
883
+ // exact count the proportional clamp needs as its denominator.
884
+ const budget = tokenizer.checkTokenBudget(text, window.budgetTokens);
885
+ try {
886
+ carriedSummary = await summarizeConversationWindow(
887
+ budget.fits ? text : clampConversationToBudget(text, window.budgetTokens, budget.tokens),
888
+ carriedSummary,
889
+ model,
890
+ maxTokens,
891
+ apiKey,
892
+ signal,
893
+ customInstructions,
894
+ options,
895
+ );
896
+ } catch (error) {
897
+ // The catalog window can overstate what the provider actually accepts:
898
+ // `claude-sonnet-4-5` advertises 1M but is beta-gated to 200k on OAuth
899
+ // credentials (see `anthropic.ts` — the 1M beta is never advertised).
900
+ // Halve and re-plan rather than failing the whole compaction on a
901
+ // window size only the provider can tell us is wrong.
902
+ // Halve what was actually SENT, not the budget it was planned against:
903
+ // the rejection proves the plan was fiction, so converging on the real
904
+ // cap must not spend a call per level of an imaginary ladder. The cheap
905
+ // fit path never counted this window, so pay for the exact size here —
906
+ // one tokenization is nothing against the provider round trip already lost.
907
+ const sentTokens = budget.exact ? budget.tokens : tokenizer.countTokens(text, "strict");
908
+ const halved = Math.floor(Math.min(window.budgetTokens, sentTokens) / 2);
909
+ if (
910
+ !AIError.is(AIError.classify(error), AIError.Flag.ContextOverflow) ||
911
+ halved < minSummaryInputTokens(model)
912
+ ) {
913
+ throw error;
914
+ }
915
+ pending.splice(
916
+ 0,
917
+ 1,
918
+ ...planSummaryWindows(window.messages, tokenizer, dialect, halved).map(messages => ({
919
+ messages,
920
+ budgetTokens: halved,
921
+ })),
922
+ );
923
+ continue;
924
+ }
925
+ pending.shift();
926
+ }
927
+ return carriedSummary ?? "";
928
+ }
929
+
930
+ /** One summarization call over a single conversation window. */
931
+ async function summarizeConversationWindow(
932
+ conversationText: string,
933
+ previousSummary: string | undefined,
934
+ model: Model,
935
+ maxTokens: number,
936
+ apiKey: ApiKey,
937
+ signal: AbortSignal | undefined,
938
+ customInstructions: string | undefined,
939
+ options: SummaryOptions | undefined,
940
+ ): Promise<string> {
902
941
  // Use update prompt if we have a previous summary, otherwise initial prompt
903
942
  let basePrompt = previousSummary ? UPDATE_SUMMARIZATION_PROMPT : SUMMARIZATION_PROMPT;
904
943
  if (options?.promptOverride) {
@@ -908,15 +947,10 @@ export async function generateSummary(
908
947
  basePrompt = `${basePrompt}\n\nAdditional focus: ${customInstructions}`;
909
948
  }
910
949
 
911
- // Serialize conversation to text so model doesn't try to continue it
912
- // Convert to LLM messages first (handles custom app messages when caller provides a transformer).
913
- const llmMessages = (options?.convertToLlm ?? defaultConvertToLlm)(currentMessages);
914
- const conversationText = serializeConversationForSummary(llmMessages, preferredDialect(model.id));
915
-
916
950
  // Build the prompt with conversation wrapped in tags
917
951
  let promptText = `<conversation>\n${conversationText}\n</conversation>\n\n`;
918
952
  if (previousSummary) {
919
- promptText += `<previous-summary>\n${previousSummary}\n</previous-summary>\n\n`;
953
+ promptText += `<previous-summary>\n${escapeSummaryBoundaryTags(previousSummary)}\n</previous-summary>\n\n`;
920
954
  }
921
955
  promptText += formatAdditionalContext(options?.extraContext);
922
956
  promptText += basePrompt;
@@ -1135,7 +1169,7 @@ async function generateShortSummary(
1135
1169
 
1136
1170
  let promptText = `<conversation>\n${conversationText}\n</conversation>\n\n`;
1137
1171
  if (historySummary) {
1138
- promptText += `<previous-summary>\n${historySummary}\n</previous-summary>\n\n`;
1172
+ promptText += `<previous-summary>\n${escapeSummaryBoundaryTags(historySummary)}\n</previous-summary>\n\n`;
1139
1173
  }
1140
1174
  promptText += formatAdditionalContext(options?.extraContext);
1141
1175
  promptText += SHORT_SUMMARY_PROMPT;
@@ -1234,7 +1268,7 @@ export interface CompactionPreparation {
1234
1268
  * let the active model replay it, so keying reuse on "any candidate shares the
1235
1269
  * provider" left a provider-switched session permanently context-less (#6343).
1236
1270
  */
1237
- function remotePreserveReusable(
1271
+ export function remotePreserveReusable(
1238
1272
  preserveData: Record<string, unknown> | undefined,
1239
1273
  activeModel: Model,
1240
1274
  settings: CompactionSettings,
@@ -1247,38 +1281,75 @@ function remotePreserveReusable(
1247
1281
  return v2Ok || shouldUseOpenAiRemoteCompaction(activeModel);
1248
1282
  }
1249
1283
 
1284
+ /**
1285
+ * Index of the newest compaction entry the active model can actually read, or
1286
+ * `-1` when none can.
1287
+ *
1288
+ * A provider-native remote compaction (V2 or V1) stores an opaque replay payload
1289
+ * and only a placeholder summary, so for any OTHER provider that entry
1290
+ * summarizes nothing and the history behind it is still live context. Callers
1291
+ * must therefore treat it as absent: `prepareCompaction` re-expands past it and
1292
+ * summarizes those messages locally, and the maintenance ops that use the
1293
+ * compaction boundary to skip "already summarized away" entries must not skip
1294
+ * entries that no summary covers.
1295
+ */
1296
+ export function findReadableCompactionIndex(
1297
+ pathEntries: SessionEntry[],
1298
+ settings: CompactionSettings,
1299
+ activeModel?: Model,
1300
+ ): number {
1301
+ for (let i = pathEntries.length - 1; i >= 0; i--) {
1302
+ if (pathEntries[i].type !== "compaction") continue;
1303
+ const entry = pathEntries[i] as CompactionEntry;
1304
+ if (activeModel && !remotePreserveReusable(entry.preserveData, activeModel, settings)) continue;
1305
+ return i;
1306
+ }
1307
+ return -1;
1308
+ }
1309
+
1310
+ /**
1311
+ * Pass the caller's warm `tokenizer` (the Agent's for the active model) so the
1312
+ * full-branch estimate walk hits its memo; the cold default is for one-shot
1313
+ * callers that have no live agent.
1314
+ */
1250
1315
  export function prepareCompaction(
1251
1316
  pathEntries: SessionEntry[],
1252
1317
  settings: CompactionSettings,
1253
1318
  activeModel?: Model,
1319
+ tokenizer: Tokenizer = new Tokenizer(activeModel),
1254
1320
  ): CompactionPreparation | undefined {
1255
1321
  if (pathEntries.length > 0 && pathEntries[pathEntries.length - 1].type === "compaction") {
1256
1322
  return undefined;
1257
1323
  }
1258
1324
 
1259
- let prevCompactionIndex = -1;
1260
- for (let i = pathEntries.length - 1; i >= 0; i--) {
1261
- if (pathEntries[i].type !== "compaction") continue;
1262
- // Skip a prior remote compaction (V2 or V1) whose provider-native replay the
1263
- // active model cannot read: its summary is only an opaque placeholder, so
1264
- // re-expand its original messages and summarize them locally rather than
1265
- // stranding that history. compact() still reuses the payload when the active
1266
- // model can replay it (same provider, remote enabled).
1267
- const entry = pathEntries[i] as CompactionEntry;
1268
- if (activeModel && !remotePreserveReusable(entry.preserveData, activeModel, settings)) {
1269
- continue;
1325
+ let prevCompactionIndex = findReadableCompactionIndex(pathEntries, settings, activeModel);
1326
+
1327
+ // Honor the latest `/clear` reset boundary. `/clear` records a
1328
+ // `reset_boundary` marker and reports the model context empty, so compaction
1329
+ // must not resurrect the dropped pre-clear turns into its summary — matching
1330
+ // how buildSessionContext starts the model-context rebuild after the boundary.
1331
+ // A boundary after the last reusable compaction supersedes it: the pre-reset
1332
+ // summary was cleared too, so drop the previous-compaction reuse and start
1333
+ // fresh after the boundary. A boundary at or before that compaction is already
1334
+ // superseded by it, so only scan newer entries.
1335
+ let resetBoundaryIndex = -1;
1336
+ for (let i = pathEntries.length - 1; i > prevCompactionIndex; i--) {
1337
+ if (pathEntries[i].type === "reset_boundary") {
1338
+ resetBoundaryIndex = i;
1339
+ break;
1270
1340
  }
1271
- prevCompactionIndex = i;
1272
- break;
1273
1341
  }
1274
- const boundaryStart = prevCompactionIndex + 1;
1342
+ if (resetBoundaryIndex > prevCompactionIndex) {
1343
+ prevCompactionIndex = -1;
1344
+ }
1345
+ const boundaryStart = Math.max(prevCompactionIndex, resetBoundaryIndex) + 1;
1275
1346
  const boundaryEnd = pathEntries.length;
1276
1347
 
1277
1348
  const lastUsage = getLastAssistantUsage(pathEntries);
1278
1349
  const tokensBefore = lastUsage ? calculateContextTokens(lastUsage) : 0;
1279
1350
  let keepRecentTokens = settings.keepRecentTokens;
1280
1351
  if (lastUsage) {
1281
- const estimatedTokens = estimateEntriesTokens(pathEntries, boundaryStart, boundaryEnd);
1352
+ const estimatedTokens = estimateEntriesTokens(pathEntries, tokenizer, boundaryStart, boundaryEnd);
1282
1353
  const promptTokens = calculatePromptTokens(lastUsage);
1283
1354
  const ratio = estimatedTokens > 0 ? promptTokens / estimatedTokens : 0;
1284
1355
  if (Number.isFinite(ratio) && ratio > 1) {
@@ -1286,7 +1357,7 @@ export function prepareCompaction(
1286
1357
  }
1287
1358
  }
1288
1359
 
1289
- const cutPoint = findCutPoint(pathEntries, boundaryStart, boundaryEnd, keepRecentTokens);
1360
+ const cutPoint = findCutPoint(pathEntries, tokenizer, boundaryStart, boundaryEnd, keepRecentTokens);
1290
1361
 
1291
1362
  // Get ID of first kept entry
1292
1363
  const firstKeptEntry = pathEntries[cutPoint.firstKeptEntryIndex];
@@ -1525,6 +1596,7 @@ export async function compact(
1525
1596
  : undefined;
1526
1597
  const trimmed = trimRemoteCompactionInputToContextWindow(
1527
1598
  remoteHistory,
1599
+ new Tokenizer(model),
1528
1600
  model.contextWindow,
1529
1601
  instructions,
1530
1602
  tools,
@@ -117,6 +117,16 @@ export interface ModeChangeEntry extends SessionEntryBase {
117
117
  data?: Record<string, unknown>;
118
118
  }
119
119
 
120
+ /**
121
+ * Durable context-reset marker recorded by an in-place `/clear`. It carries no
122
+ * payload — its presence on the branch means every entry before it was dropped
123
+ * from the model context, so context assembly and compaction start after the
124
+ * latest one. The full pre-reset history stays on disk for transcript export.
125
+ */
126
+ export interface ResetBoundaryEntry extends SessionEntryBase {
127
+ type: "reset_boundary";
128
+ }
129
+
120
130
  export interface CustomCompactionSessionEntries {}
121
131
 
122
132
  export type SessionEntry =
@@ -133,6 +143,7 @@ export type SessionEntry =
133
143
  | TtsrInjectionEntry
134
144
  | SessionInitEntry
135
145
  | ModeChangeEntry
146
+ | ResetBoundaryEntry
136
147
  | CustomCompactionSessionEntries[keyof CustomCompactionSessionEntries];
137
148
 
138
149
  export interface ReadonlySessionManager {
@@ -11,4 +11,5 @@ export * from "./messages";
11
11
  export * from "./openai";
12
12
  export * from "./pruning";
13
13
  export * from "./shake";
14
+ export * from "./transcript-tokens";
14
15
  export * from "./utils";