@oh-my-pi/pi-agent-core 18.1.16 → 18.1.18

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,29 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.1.18] - 2026-09-11
6
+
7
+ ### Added
8
+
9
+ - Anthropic server-side compaction as a `remote` compaction backend: model lines the beta supports (`compat.supportsServerCompaction`, rule-owned in the catalog: Opus 4.6+, Sonnet 4.6+, Fable/Mythos 5) on the official endpoint, resolved the way the provider routes requests, plus Anthropic-compatible routes with `remoteCompaction.enabled`, compact by re-issuing the live turn's own request — same system prompt, tools, and history, so it reads the prompt cache the last turn wrote — with the `compact_20260112` edit paused after the summary and the harness summary prompt as `instructions`. The instructions name where the retained tail begins so the summary covers only the history the rebuilt context drops. The API's summary is stored as the entry text and as `preserveData.anthropicCompaction`, replayed natively on later Anthropic requests and read as plain text by every other provider; the retained tail comes from session entries as with a local summary. Contexts below 55k tokens (the API trigger floor plus margin) keep summarizing locally, and a response without a summary is a native failure, like the OpenAI lanes. An aborted compaction response is the abort (a cancellation, never a native failure) and an error response keeps its HTTP status, so auth and timeout classification match the OpenAI lanes; the block's opaque `encrypted_content` is persisted as `preserveData.anthropicCompaction.encryptedContent` and replayed verbatim.
10
+
11
+ ### Fixed
12
+
13
+ - `compact()` now forwards the caller's `oneshotRetry` opt-out to every summarization oneshot; auto-compaction's outer retry loop no longer multiplies with the inner transient-failure retries.
14
+
15
+ ## [18.1.17] - 2026-09-10
16
+
17
+ ### Changed
18
+
19
+ - `Tool <name> not found` now names a plausible intended target when the advertised set contains one, e.g. `Tool mcp__abc123__xyz789_read not found. Did you mean read?`. A model that mis-transcribes a long opaque tool name reliably keeps the trailing segment, which is the only part carrying meaning, so the miss becomes recoverable in the same turn instead of costing a round trip. Purely advisory — the suggestion is only ever a string in the error, never a dispatch target, so an unrecognized name still fails ([#10109](https://github.com/can1357/oh-my-pi/issues/10109) by [@oldschoola](https://github.com/oldschoola)).
20
+
21
+ ### Fixed
22
+
23
+ - Fixed the token estimator counting developer messages as free and ignoring images in user content, which let context budgeting, pruning and the compaction trigger read a transcript as far smaller than the one sent to the provider.
24
+ - Fixed repeated local compaction omitting messages retained before the previous compaction record, while preserving original entry IDs and `/clear` boundaries.
25
+ - Raised remote compaction request timeout from 3 minutes to 5 minutes so long Codex/gpt-6-astra compact streams can finish before the watchdog aborts them.
26
+ - Fixed proxy responses dropping the cost the server reported; recorded costs are kept instead of being recomputed.
27
+
5
28
  ## [18.1.10] - 2026-09-04
6
29
 
7
30
  ### Fixed
@@ -0,0 +1,108 @@
1
+ /**
2
+ * Anthropic server-side compaction (`compact-2026-01-12` beta).
3
+ *
4
+ * The compaction request is the live turn's own request shape — same system
5
+ * prompt, tools, and message history — plus the `compact_20260112` edit with
6
+ * `pause_after_compaction`. The API summarizes the prompt from the already
7
+ * cached prefix and stops; the summary arrives as a `compaction` block that
8
+ * the provider surfaces as an `anthropicCompaction` payload. The summary is
9
+ * plain text, so it doubles as the compaction entry's readable summary for
10
+ * every other provider, while the Anthropic provider replays it as a native
11
+ * block (the API drops everything that precedes it). The retained tail after
12
+ * the cut point is replayed from session entries exactly like a local summary.
13
+ */
14
+ import type { AnthropicCompactionPayload, ApiKey, Effort, Message, Model, SimpleStreamOptions, Tool, Usage } from "@oh-my-pi/pi-ai";
15
+ import { type InstrumentedChatSpanOptions } from "../telemetry.js";
16
+ export declare const ANTHROPIC_COMPACTION_PRESERVE_KEY = "anthropicCompaction";
17
+ /** The API rejects a `compact_20260112` trigger below this many input tokens. */
18
+ export declare const ANTHROPIC_COMPACTION_MIN_TRIGGER_TOKENS = 50000;
19
+ /**
20
+ * Smallest context the native lane accepts. The trigger sits at the API
21
+ * floor, so a prompt that lands below it is answered instead of compacted;
22
+ * the margin over the floor absorbs the difference between the last reported
23
+ * context size and the compaction request's own input.
24
+ */
25
+ export declare const ANTHROPIC_COMPACTION_MIN_CONTEXT_TOKENS = 55000;
26
+ /** Summary persisted under {@link ANTHROPIC_COMPACTION_PRESERVE_KEY}. */
27
+ export interface AnthropicCompactionPreserveData {
28
+ provider: string;
29
+ content: string;
30
+ /** Opaque provider state the API attached to the block; replayed verbatim. */
31
+ encryptedContent?: string;
32
+ /** Harness file metadata (`<files>` section) replayed after the native block. */
33
+ filesText?: string;
34
+ /** Model that wrote the summary. */
35
+ model?: string;
36
+ /** Prompt tokens the compaction request processed, for display. */
37
+ usedTokens?: number;
38
+ }
39
+ /**
40
+ * Whether a model compacts through the Anthropic compaction beta. Model
41
+ * eligibility is catalog policy (`compat.supportsServerCompaction`, the
42
+ * lineage the beta documents); endpoint eligibility is resolved the way the
43
+ * provider routes requests, so a Foundry or `ANTHROPIC_BASE_URL` reroute of a
44
+ * first-party model is excluded unless the route opted in with
45
+ * `remoteCompaction.enabled`.
46
+ */
47
+ export declare function shouldUseAnthropicNativeCompaction(model: Model): model is Model<"anthropic-messages">;
48
+ export declare function getPreservedAnthropicCompactionData(preserveData: Record<string, unknown> | undefined): AnthropicCompactionPreserveData | undefined;
49
+ /** Set or strip the Anthropic compaction slot; a new compaction never inherits a stale summary. */
50
+ export declare function withAnthropicCompactionPreserveData(preserveData: Record<string, unknown> | undefined, compaction: AnthropicCompactionPreserveData | undefined): Record<string, unknown> | undefined;
51
+ /** Replay payload for a compaction summary the active model produced natively. */
52
+ export declare function getAnthropicCompactionPayload(preserveData: Record<string, unknown> | undefined): AnthropicCompactionPayload | undefined;
53
+ /**
54
+ * The retained tail as the model will see it, for the summarization
55
+ * instructions: how many of the conversation's final wire messages stay in
56
+ * context verbatim, and the role of the first. The compaction request carries
57
+ * the whole conversation so the prompt cache the live turn wrote is read, but
58
+ * the summary must cover only the history before that tail — the local
59
+ * summarizer never sees the tail, and the rebuilt context replays it after the
60
+ * summary. Counting mirrors the provider's message conversion (consecutive
61
+ * tool results collapse into one user message; developer messages are user
62
+ * messages). Structured for the prompt template, which renders the
63
+ * singular/plural wording; the description quotes no content: quoting the
64
+ * tail would hand the summarizer the very facts it must leave to the tail.
65
+ */
66
+ export interface RetainedTailScope {
67
+ count: number;
68
+ role: "assistant" | "user";
69
+ }
70
+ export declare function describeRetainedTail(messages: readonly Message[]): RetainedTailScope | undefined;
71
+ /**
72
+ * Summarization prompt sent as the edit's `instructions`, which replace the
73
+ * API default entirely. The template lays out the retained-tail boundary
74
+ * first, so the summary covers only the history the rebuilt context drops,
75
+ * then the caller's extra context, the same structure prompt as the local
76
+ * summarizer, the caller's focus, and the tool-abstention clause the API
77
+ * recommends when tools are defined (a summarization pass that calls a tool
78
+ * yields no summary).
79
+ */
80
+ export declare function buildAnthropicCompactionInstructions(basePrompt: string, customInstructions: string | undefined, extraContext: string | undefined, retainedTail: RetainedTailScope | undefined): string;
81
+ export interface AnthropicNativeCompactionRequest {
82
+ systemPrompt: string[];
83
+ messages: Message[];
84
+ tools?: Tool[];
85
+ instructions: string;
86
+ maxTokens: number;
87
+ reasoning?: Effort;
88
+ }
89
+ export interface AnthropicNativeCompactionResponse {
90
+ content: string;
91
+ encryptedContent?: string;
92
+ usage: Usage;
93
+ model: string;
94
+ }
95
+ export interface AnthropicNativeCompactionOptions extends Pick<SimpleStreamOptions, "initiatorOverride" | "metadata" | "fetch" | "sessionId" | "promptCacheKey" | "providerSessionState" | "maxInFlightRequests">, Pick<InstrumentedChatSpanOptions, "completeImpl" | "telemetry" | "retry"> {
96
+ }
97
+ /**
98
+ * Run one compaction request and return the summary the API wrote, with the
99
+ * opaque `encrypted_content` the API attached for the replay. `completeSimple`
100
+ * resolves terminal failures as messages, so their classification is restored
101
+ * here: an aborted response is an `AbortError` (a cancellation, never a native
102
+ * failure) and an error response keeps its HTTP status, so auth and timeout
103
+ * handling downstream classify it the same way as the OpenAI lanes. A response
104
+ * without a summary is a native failure — the API answers the prompt instead
105
+ * when its input never reached the trigger, and returns an empty block when
106
+ * the model called a tool during summarization.
107
+ */
108
+ export declare function requestAnthropicNativeCompaction(model: Model<"anthropic-messages">, apiKey: ApiKey, request: AnthropicNativeCompactionRequest, signal: AbortSignal | undefined, options: AnthropicNativeCompactionOptions): Promise<AnthropicNativeCompactionResponse>;
@@ -12,8 +12,8 @@ import { type OpenAICodexCompactionBody } from "@oh-my-pi/pi-ai/providers/openai
12
12
  export declare const V2_RETAINED_MESSAGE_TOKEN_BUDGET = 64000;
13
13
  /** Max retries for V2 streaming compaction on transient stream errors. */
14
14
  export declare const V2_COMPACTION_MAX_RETRIES = 2;
15
- /** Timeout for V2 streaming compaction (3 minutes, same as V1). */
16
- export declare const V2_COMPACTION_TIMEOUT_MS = 180000;
15
+ /** Timeout for V2 streaming compaction (5 minutes, same as V1). */
16
+ export declare const V2_COMPACTION_TIMEOUT_MS = 300000;
17
17
  /** Token usage reported by the streamed V2 Responses completion. */
18
18
  export interface CompactionV2Usage {
19
19
  inputTokens: number;
@@ -65,7 +65,11 @@ export declare const DEFAULT_RESERVE_TOKENS = 16384;
65
65
  */
66
66
  export declare const MAX_SUMMARY_TOKENS = 16384;
67
67
  export declare const DEFAULT_COMPACTION_SETTINGS: CompactionSettings;
68
- /** Whether a compaction candidate preserves provider-native transport under the effective settings. */
68
+ /**
69
+ * Whether a compaction candidate preserves provider-native transport under the
70
+ * effective settings: an OpenAI Responses compact route (V1 or streamed V2) or
71
+ * the Anthropic compaction beta.
72
+ */
69
73
  export declare function shouldUseProviderNativeCompaction(model: Model, settings: Pick<CompactionSettings, "remoteEnabled" | "remoteStreamingV2Enabled">): boolean;
70
74
  /**
71
75
  * Calculate total context tokens from usage.
@@ -290,6 +294,8 @@ export interface CompactionPreparation {
290
294
  tokensBefore: number;
291
295
  /** Summary from previous compaction, for iterative update */
292
296
  previousSummary?: string;
297
+ /** ISO timestamp of the previous compaction entry, for iterative update */
298
+ previousSummaryTimestamp?: string;
293
299
  /** Preserved opaque compaction payload from the previous compaction, if any. */
294
300
  previousPreserveData?: Record<string, unknown>;
295
301
  /** File operations extracted from messagesToSummarize */
@@ -1,6 +1,7 @@
1
1
  /**
2
2
  * Compaction and summarization utilities.
3
3
  */
4
+ export * from "./anthropic.js";
4
5
  export * from "./branch-summarization.js";
5
6
  export * from "./compaction.js";
6
7
  export * from "./entries.js";
@@ -19,14 +19,14 @@ import { Tokenizer } from "../tokenizer.js";
19
19
  export * from "./compaction-v2-streaming.js";
20
20
  export declare const OPENAI_REMOTE_COMPACTION_PRESERVE_KEY = "openaiRemoteCompaction";
21
21
  /**
22
- * Hard ceiling on remote compaction HTTP requests. Unlike every provider
23
- * stream (guarded by first-event/idle watchdogs in pi-ai), these are raw
24
- * fetches awaiting one non-streamed JSON body — a connection silently dropped
25
- * by a middlebox would otherwise hang the whole compaction pipeline forever
26
- * (frozen "Auto context-full maintenance…" spinner, manual /compact queueing
27
- * behind it). On timeout the caller falls back to local summarization.
22
+ * Hard ceiling on remote compaction HTTP requests (5 minutes). Unlike every
23
+ * provider stream (guarded by first-event/idle watchdogs in pi-ai), these are
24
+ * raw fetches awaiting one non-streamed JSON body — a connection silently
25
+ * dropped by a middlebox would otherwise hang the whole compaction pipeline
26
+ * forever (frozen "Auto context-full maintenance…" spinner, manual /compact
27
+ * queueing behind it). On timeout the caller falls back to local summarization.
28
28
  */
29
- export declare const REMOTE_COMPACTION_TIMEOUT_MS = 180000;
29
+ export declare const REMOTE_COMPACTION_TIMEOUT_MS = 300000;
30
30
  export declare const CONTEXT_WINDOW_TRUNCATED_OUTPUT_MESSAGE: string;
31
31
  export interface TrimRemoteCompactionInputResult {
32
32
  input: Array<Record<string, unknown>>;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@oh-my-pi/pi-agent-core",
4
- "version": "18.1.16",
4
+ "version": "18.1.18",
5
5
  "description": "General-purpose agent with transport abstraction, state management, and attachment support",
6
6
  "homepage": "https://omp.sh",
7
7
  "author": "Stencil Labs, Inc.",
@@ -35,16 +35,16 @@
35
35
  "fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
36
36
  },
37
37
  "dependencies": {
38
- "@oh-my-pi/pi-ai": "18.1.16",
39
- "@oh-my-pi/pi-catalog": "18.1.16",
40
- "@oh-my-pi/pi-natives": "18.1.16",
41
- "@oh-my-pi/pi-utils": "18.1.16",
42
- "@oh-my-pi/pi-wire": "18.1.16",
43
- "@oh-my-pi/snapcompact": "18.1.16",
38
+ "@oh-my-pi/pi-ai": "18.1.18",
39
+ "@oh-my-pi/pi-catalog": "18.1.18",
40
+ "@oh-my-pi/pi-natives": "18.1.18",
41
+ "@oh-my-pi/pi-utils": "18.1.18",
42
+ "@oh-my-pi/pi-wire": "18.1.18",
43
+ "@oh-my-pi/snapcompact": "18.1.18",
44
44
  "@opentelemetry/api": "^1.9.1"
45
45
  },
46
46
  "devDependencies": {
47
- "@oh-my-pi/omptype": "18.1.16",
47
+ "@oh-my-pi/omptype": "18.1.18",
48
48
  "@opentelemetry/context-async-hooks": "^2.9.0",
49
49
  "@opentelemetry/sdk-trace-base": "^2.9.0",
50
50
  "@types/bun": "^1.3.14"
package/src/agent-loop.ts CHANGED
@@ -2269,6 +2269,68 @@ function resolveToolForCall(
2269
2269
  );
2270
2270
  }
2271
2271
 
2272
+ /** Shortest suggestable segment; below this the match is noise (`id`, `to`). */
2273
+ const MIN_TOOL_NAME_SUGGESTION_SEGMENT = 3;
2274
+ /** Cap on names listed for an ambiguous miss, so the error stays readable. */
2275
+ const MAX_TOOL_NAME_SUGGESTIONS = 3;
2276
+
2277
+ /**
2278
+ * Advertised tool names sharing a trailing `_`-delimited segment with `name`.
2279
+ *
2280
+ * A model that mis-transcribes a long opaque tool name reliably keeps the
2281
+ * trailing verb — that segment is the only part carrying meaning, while any
2282
+ * leading id segments are high-entropy and mnemonic-free. Matching on it turns
2283
+ * an otherwise dead `not found` into a self-correcting one.
2284
+ *
2285
+ * Both the last `__` and last `_` boundary are tried, so a name that lost only
2286
+ * its separator (`…__resolve_library_id`) and one that lost a whole id segment
2287
+ * (`…__read`) both recover. Purely advisory: this only builds an error string
2288
+ * and never selects a tool, so dispatch semantics are unchanged.
2289
+ */
2290
+ function suggestToolNames(
2291
+ name: string,
2292
+ tools: ReadonlyArray<Pick<AgentTool, "name" | "customWireName">> | undefined,
2293
+ ): string[] {
2294
+ if (!tools || tools.length === 0) return [];
2295
+ const segments: string[] = [];
2296
+ for (const boundary of ["__", "_"]) {
2297
+ const idx = name.lastIndexOf(boundary);
2298
+ if (idx < 0) continue;
2299
+ const segment = name.slice(idx + boundary.length);
2300
+ if (segment.length >= MIN_TOOL_NAME_SUGGESTION_SEGMENT && !segments.includes(segment)) segments.push(segment);
2301
+ }
2302
+ if (segments.length === 0) return [];
2303
+ // Longest tail first. A distinctive `__` tail (`resolve_library_get`) is a
2304
+ // far stronger signal than the generic `_` tail it contains (`get`), and the
2305
+ // caller truncates the list — so the strongest match has to sort ahead of
2306
+ // however many tools happen to share the weak one.
2307
+ segments.sort((a, b) => b.length - a.length);
2308
+ const matches: string[] = [];
2309
+ for (const segment of segments) {
2310
+ for (const tool of tools) {
2311
+ for (const candidate of [tool.name, tool.customWireName]) {
2312
+ if (candidate === undefined || candidate === name || matches.includes(candidate)) continue;
2313
+ if (candidate === segment || candidate.endsWith(`_${segment}`)) matches.push(candidate);
2314
+ }
2315
+ }
2316
+ }
2317
+ return matches;
2318
+ }
2319
+
2320
+ /**
2321
+ * `Tool <name> not found`, plus a suggestion when the advertised set contains a
2322
+ * plausible intended target. Exact wording is not a contract; the model reads it.
2323
+ */
2324
+ function formatToolNotFoundMessage(
2325
+ name: string,
2326
+ tools: ReadonlyArray<Pick<AgentTool, "name" | "customWireName">> | undefined,
2327
+ ): string {
2328
+ const suggestions = suggestToolNames(name, tools);
2329
+ if (suggestions.length === 0) return `Tool ${name} not found`;
2330
+ if (suggestions.length === 1) return `Tool ${name} not found. Did you mean ${suggestions[0]}?`;
2331
+ return `Tool ${name} not found. Closest available: ${suggestions.slice(0, MAX_TOOL_NAME_SUGGESTIONS).join(", ")}`;
2332
+ }
2333
+
2272
2334
  /**
2273
2335
  * Pre-dispatch phase for every pending tool call on `assistantMessage`, run in
2274
2336
  * call order: intent extraction, argument validation, and the `beforeToolCall`
@@ -2312,7 +2374,7 @@ async function prepareToolCallDispatch(
2312
2374
  }
2313
2375
  const validate = (args: Record<string, unknown>): Record<string, unknown> | undefined => {
2314
2376
  try {
2315
- if (!tool) throw new Error(`Tool ${toolCall.name} not found`);
2377
+ if (!tool) throw new Error(formatToolNotFoundMessage(toolCall.name, context.tools));
2316
2378
  return validateToolArguments(tool, { ...toolCall, arguments: args });
2317
2379
  } catch (validationError) {
2318
2380
  if (tool?.lenientArgValidation) {
@@ -2637,7 +2699,7 @@ async function executeToolCalls(
2637
2699
 
2638
2700
  await runInActiveSpan(toolSpan, async () => {
2639
2701
  try {
2640
- if (!tool) throw new Error(`Tool ${toolCall.name} not found`);
2702
+ if (!tool) throw new Error(formatToolNotFoundMessage(toolCall.name, tools));
2641
2703
  if (record.signal.aborted) {
2642
2704
  result = createToolSignalAbortedResult(record.signal);
2643
2705
  isError = true;
@@ -0,0 +1,310 @@
1
+ /**
2
+ * Anthropic server-side compaction (`compact-2026-01-12` beta).
3
+ *
4
+ * The compaction request is the live turn's own request shape — same system
5
+ * prompt, tools, and message history — plus the `compact_20260112` edit with
6
+ * `pause_after_compaction`. The API summarizes the prompt from the already
7
+ * cached prefix and stops; the summary arrives as a `compaction` block that
8
+ * the provider surfaces as an `anthropicCompaction` payload. The summary is
9
+ * plain text, so it doubles as the compaction entry's readable summary for
10
+ * every other provider, while the Anthropic provider replays it as a native
11
+ * block (the API drops everything that precedes it). The retained tail after
12
+ * the cut point is replayed from session entries exactly like a local summary.
13
+ */
14
+
15
+ import type {
16
+ AnthropicCompactionPayload,
17
+ ApiKey,
18
+ AssistantMessage,
19
+ Effort,
20
+ Message,
21
+ Model,
22
+ SimpleStreamOptions,
23
+ Tool,
24
+ Usage,
25
+ } from "@oh-my-pi/pi-ai";
26
+ import * as AIError from "@oh-my-pi/pi-ai/error";
27
+ import { supportsAnthropicCompaction } from "@oh-my-pi/pi-ai/providers/anthropic";
28
+ import { isRecord, prompt } from "@oh-my-pi/pi-utils";
29
+ import { type InstrumentedChatSpanOptions, instrumentedCompleteSimple } from "../telemetry";
30
+ import anthropicCompactionInstructionsPrompt from "./prompts/anthropic-compaction-instructions.md" with { type: "text" };
31
+
32
+ export const ANTHROPIC_COMPACTION_PRESERVE_KEY = "anthropicCompaction";
33
+
34
+ /** The API rejects a `compact_20260112` trigger below this many input tokens. */
35
+ export const ANTHROPIC_COMPACTION_MIN_TRIGGER_TOKENS = 50_000;
36
+
37
+ /**
38
+ * Smallest context the native lane accepts. The trigger sits at the API
39
+ * floor, so a prompt that lands below it is answered instead of compacted;
40
+ * the margin over the floor absorbs the difference between the last reported
41
+ * context size and the compaction request's own input.
42
+ */
43
+ export const ANTHROPIC_COMPACTION_MIN_CONTEXT_TOKENS = 55_000;
44
+
45
+ /** Summary persisted under {@link ANTHROPIC_COMPACTION_PRESERVE_KEY}. */
46
+ export interface AnthropicCompactionPreserveData {
47
+ provider: string;
48
+ content: string;
49
+ /** Opaque provider state the API attached to the block; replayed verbatim. */
50
+ encryptedContent?: string;
51
+ /** Harness file metadata (`<files>` section) replayed after the native block. */
52
+ filesText?: string;
53
+ /** Model that wrote the summary. */
54
+ model?: string;
55
+ /** Prompt tokens the compaction request processed, for display. */
56
+ usedTokens?: number;
57
+ }
58
+
59
+ function isAnthropicMessagesModel(model: Model): model is Model<"anthropic-messages"> {
60
+ return model.api === "anthropic-messages";
61
+ }
62
+
63
+ /**
64
+ * Whether a model compacts through the Anthropic compaction beta. Model
65
+ * eligibility is catalog policy (`compat.supportsServerCompaction`, the
66
+ * lineage the beta documents); endpoint eligibility is resolved the way the
67
+ * provider routes requests, so a Foundry or `ANTHROPIC_BASE_URL` reroute of a
68
+ * first-party model is excluded unless the route opted in with
69
+ * `remoteCompaction.enabled`.
70
+ */
71
+ export function shouldUseAnthropicNativeCompaction(model: Model): model is Model<"anthropic-messages"> {
72
+ return isAnthropicMessagesModel(model) && supportsAnthropicCompaction(model);
73
+ }
74
+
75
+ export function getPreservedAnthropicCompactionData(
76
+ preserveData: Record<string, unknown> | undefined,
77
+ ): AnthropicCompactionPreserveData | undefined {
78
+ const candidate = preserveData?.[ANTHROPIC_COMPACTION_PRESERVE_KEY];
79
+ if (!isRecord(candidate)) return undefined;
80
+ if (typeof candidate.provider !== "string" || candidate.provider.length === 0) return undefined;
81
+ if (typeof candidate.content !== "string" || candidate.content.length === 0) return undefined;
82
+ return {
83
+ provider: candidate.provider,
84
+ content: candidate.content,
85
+ ...(typeof candidate.encryptedContent === "string" && candidate.encryptedContent.length > 0
86
+ ? { encryptedContent: candidate.encryptedContent }
87
+ : {}),
88
+ ...(typeof candidate.filesText === "string" && candidate.filesText.length > 0
89
+ ? { filesText: candidate.filesText }
90
+ : {}),
91
+ ...(typeof candidate.model === "string" ? { model: candidate.model } : {}),
92
+ ...(typeof candidate.usedTokens === "number" ? { usedTokens: candidate.usedTokens } : {}),
93
+ };
94
+ }
95
+
96
+ /** Set or strip the Anthropic compaction slot; a new compaction never inherits a stale summary. */
97
+ export function withAnthropicCompactionPreserveData(
98
+ preserveData: Record<string, unknown> | undefined,
99
+ compaction: AnthropicCompactionPreserveData | undefined,
100
+ ): Record<string, unknown> | undefined {
101
+ if (compaction) {
102
+ return { ...preserveData, [ANTHROPIC_COMPACTION_PRESERVE_KEY]: compaction };
103
+ }
104
+ if (!preserveData || !(ANTHROPIC_COMPACTION_PRESERVE_KEY in preserveData)) {
105
+ return preserveData;
106
+ }
107
+ const { [ANTHROPIC_COMPACTION_PRESERVE_KEY]: _removed, ...rest } = preserveData;
108
+ return Object.keys(rest).length > 0 ? rest : undefined;
109
+ }
110
+
111
+ /** Replay payload for a compaction summary the active model produced natively. */
112
+ export function getAnthropicCompactionPayload(
113
+ preserveData: Record<string, unknown> | undefined,
114
+ ): AnthropicCompactionPayload | undefined {
115
+ const preserved = getPreservedAnthropicCompactionData(preserveData);
116
+ if (!preserved) return undefined;
117
+ return {
118
+ type: "anthropicCompaction",
119
+ provider: preserved.provider,
120
+ content: preserved.content,
121
+ ...(preserved.encryptedContent ? { encryptedContent: preserved.encryptedContent } : {}),
122
+ ...(preserved.filesText ? { filesText: preserved.filesText } : {}),
123
+ };
124
+ }
125
+
126
+ /**
127
+ * The retained tail as the model will see it, for the summarization
128
+ * instructions: how many of the conversation's final wire messages stay in
129
+ * context verbatim, and the role of the first. The compaction request carries
130
+ * the whole conversation so the prompt cache the live turn wrote is read, but
131
+ * the summary must cover only the history before that tail — the local
132
+ * summarizer never sees the tail, and the rebuilt context replays it after the
133
+ * summary. Counting mirrors the provider's message conversion (consecutive
134
+ * tool results collapse into one user message; developer messages are user
135
+ * messages). Structured for the prompt template, which renders the
136
+ * singular/plural wording; the description quotes no content: quoting the
137
+ * tail would hand the summarizer the very facts it must leave to the tail.
138
+ */
139
+ export interface RetainedTailScope {
140
+ count: number;
141
+ role: "assistant" | "user";
142
+ }
143
+
144
+ export function describeRetainedTail(messages: readonly Message[]): RetainedTailScope | undefined {
145
+ const first = messages[0];
146
+ if (!first) return undefined;
147
+ let count = 0;
148
+ let previousWasToolResult = false;
149
+ for (const message of messages) {
150
+ const isToolResult = message.role === "toolResult";
151
+ if (!(isToolResult && previousWasToolResult)) count += 1;
152
+ previousWasToolResult = isToolResult;
153
+ }
154
+ // Mirror the provider's trailing-assistant prefill: a tail ending in a
155
+ // live assistant turn gains a synthetic trailing user message on the
156
+ // wire, which stays verbatim too. Only blocks the converter emits count —
157
+ // blank text never serializes, and images, redacted thinking, and fallback
158
+ // markers need target context this scope lacks, so a turn of only those
159
+ // emits nothing and draws no pad. Without the pad in
160
+ // the scope, the summary could duplicate the tail head.
161
+ const last = messages[messages.length - 1];
162
+ if (last?.role === "assistant" && last.content.some(emitsWireBlock)) {
163
+ count += 1;
164
+ }
165
+ return { count, role: first.role === "assistant" ? "assistant" : "user" };
166
+ }
167
+
168
+ /**
169
+ * Whether an assistant content block reaches the wire. Server-tool blocks
170
+ * serialize unconditionally; blank text never does. Redacted thinking and
171
+ * fallback markers replay only for specific deployments, which the scope
172
+ * cannot see, so a turn of only those conservatively draws no pad.
173
+ */
174
+ function emitsWireBlock(block: AssistantMessage["content"][number]): boolean {
175
+ switch (block.type) {
176
+ case "text":
177
+ return block.text.trim().length > 0;
178
+ case "toolCall":
179
+ case "anthropicServerTool":
180
+ return true;
181
+ case "thinking":
182
+ return block.thinking.trim().length > 0 || (block.thinkingSignature ?? "").trim().length > 0;
183
+ default:
184
+ return false;
185
+ }
186
+ }
187
+
188
+ /**
189
+ * Summarization prompt sent as the edit's `instructions`, which replace the
190
+ * API default entirely. The template lays out the retained-tail boundary
191
+ * first, so the summary covers only the history the rebuilt context drops,
192
+ * then the caller's extra context, the same structure prompt as the local
193
+ * summarizer, the caller's focus, and the tool-abstention clause the API
194
+ * recommends when tools are defined (a summarization pass that calls a tool
195
+ * yields no summary).
196
+ */
197
+ export function buildAnthropicCompactionInstructions(
198
+ basePrompt: string,
199
+ customInstructions: string | undefined,
200
+ extraContext: string | undefined,
201
+ retainedTail: RetainedTailScope | undefined,
202
+ ): string {
203
+ return prompt.render(anthropicCompactionInstructionsPrompt, {
204
+ basePrompt,
205
+ customInstructions,
206
+ extraContext,
207
+ retainedTail,
208
+ });
209
+ }
210
+
211
+ export interface AnthropicNativeCompactionRequest {
212
+ systemPrompt: string[];
213
+ messages: Message[];
214
+ tools?: Tool[];
215
+ instructions: string;
216
+ maxTokens: number;
217
+ reasoning?: Effort;
218
+ }
219
+
220
+ export interface AnthropicNativeCompactionResponse {
221
+ content: string;
222
+ encryptedContent?: string;
223
+ usage: Usage;
224
+ model: string;
225
+ }
226
+
227
+ export interface AnthropicNativeCompactionOptions
228
+ extends
229
+ Pick<
230
+ SimpleStreamOptions,
231
+ | "initiatorOverride"
232
+ | "metadata"
233
+ | "fetch"
234
+ | "sessionId"
235
+ | "promptCacheKey"
236
+ | "providerSessionState"
237
+ | "maxInFlightRequests"
238
+ >,
239
+ Pick<InstrumentedChatSpanOptions, "completeImpl" | "telemetry" | "retry"> {}
240
+
241
+ /**
242
+ * Run one compaction request and return the summary the API wrote, with the
243
+ * opaque `encrypted_content` the API attached for the replay. `completeSimple`
244
+ * resolves terminal failures as messages, so their classification is restored
245
+ * here: an aborted response is an `AbortError` (a cancellation, never a native
246
+ * failure) and an error response keeps its HTTP status, so auth and timeout
247
+ * handling downstream classify it the same way as the OpenAI lanes. A response
248
+ * without a summary is a native failure — the API answers the prompt instead
249
+ * when its input never reached the trigger, and returns an empty block when
250
+ * the model called a tool during summarization.
251
+ */
252
+ export async function requestAnthropicNativeCompaction(
253
+ model: Model<"anthropic-messages">,
254
+ apiKey: ApiKey,
255
+ request: AnthropicNativeCompactionRequest,
256
+ signal: AbortSignal | undefined,
257
+ options: AnthropicNativeCompactionOptions,
258
+ ): Promise<AnthropicNativeCompactionResponse> {
259
+ const response = await instrumentedCompleteSimple(
260
+ model,
261
+ { systemPrompt: request.systemPrompt, messages: request.messages, tools: request.tools },
262
+ {
263
+ apiKey,
264
+ signal,
265
+ maxTokens: request.maxTokens,
266
+ reasoning: request.reasoning,
267
+ initiatorOverride: options.initiatorOverride,
268
+ metadata: options.metadata,
269
+ fetch: options.fetch,
270
+ sessionId: options.sessionId,
271
+ promptCacheKey: options.promptCacheKey,
272
+ providerSessionState: options.providerSessionState,
273
+ maxInFlightRequests: options.maxInFlightRequests,
274
+ anthropicCompaction: {
275
+ triggerInputTokens: ANTHROPIC_COMPACTION_MIN_TRIGGER_TOKENS,
276
+ pauseAfterCompaction: true,
277
+ instructions: request.instructions,
278
+ },
279
+ },
280
+ {
281
+ telemetry: options.telemetry,
282
+ oneshotKind: "compaction_native",
283
+ completeImpl: options.completeImpl,
284
+ retry: options.retry,
285
+ },
286
+ );
287
+ if (response.stopReason === "aborted") {
288
+ throw new AIError.AbortError("Anthropic compaction aborted", { cause: signal?.reason });
289
+ }
290
+ if (response.stopReason === "error") {
291
+ const message = `Anthropic compaction failed: ${response.errorMessage ?? "unknown error"}`;
292
+ throw response.errorStatus === undefined
293
+ ? new Error(message)
294
+ : new AIError.ProviderHttpError(message, response.errorStatus);
295
+ }
296
+ const payload = response.providerPayload;
297
+ if (payload?.type !== "anthropicCompaction" || payload.content.length === 0) {
298
+ throw new Error(
299
+ response.stopDetails?.type === "compaction"
300
+ ? "Anthropic compaction returned no summary"
301
+ : "Anthropic compaction response carried no compaction block",
302
+ );
303
+ }
304
+ return {
305
+ content: payload.content,
306
+ encryptedContent: payload.encryptedContent,
307
+ usage: response.usage,
308
+ model: response.model,
309
+ };
310
+ }
@@ -44,8 +44,8 @@ export const V2_RETAINED_MESSAGE_TOKEN_BUDGET = 64_000;
44
44
  /** Max retries for V2 streaming compaction on transient stream errors. */
45
45
  export const V2_COMPACTION_MAX_RETRIES = 2;
46
46
 
47
- /** Timeout for V2 streaming compaction (3 minutes, same as V1). */
48
- export const V2_COMPACTION_TIMEOUT_MS = 180_000;
47
+ /** Timeout for V2 streaming compaction (5 minutes, same as V1). */
48
+ export const V2_COMPACTION_TIMEOUT_MS = 300_000;
49
49
 
50
50
  const DEFAULT_AZURE_API_VERSION = "v1";
51
51
  const OPENAI_REMOTE_COMPACTION_PRESERVE_KEY = "openaiRemoteCompaction";
@@ -42,6 +42,15 @@ import { type AgentTelemetry, instrumentedCompleteSimple } from "../telemetry";
42
42
  import { ThinkingLevel } from "../thinking";
43
43
  import { Tokenizer } from "../tokenizer";
44
44
  import type { AgentMessage } from "../types";
45
+ import {
46
+ ANTHROPIC_COMPACTION_MIN_CONTEXT_TOKENS,
47
+ buildAnthropicCompactionInstructions,
48
+ describeRetainedTail,
49
+ getPreservedAnthropicCompactionData,
50
+ requestAnthropicNativeCompaction,
51
+ shouldUseAnthropicNativeCompaction,
52
+ withAnthropicCompactionPreserveData,
53
+ } from "./anthropic";
45
54
  import {
46
55
  buildCompactionV2Request,
47
56
  buildCompactionV2RequestFromBody,
@@ -53,7 +62,13 @@ import {
53
62
  } from "./compaction-v2-streaming";
54
63
  import type { CompactionEntry, SessionEntry } from "./entries";
55
64
  import { NativeCompactionError } from "./errors";
56
- import { type ConvertToLlm, createBranchSummaryMessage, createCustomMessage, defaultConvertToLlm } from "./messages";
65
+ import {
66
+ type ConvertToLlm,
67
+ createBranchSummaryMessage,
68
+ createCompactionSummaryMessage,
69
+ createCustomMessage,
70
+ defaultConvertToLlm,
71
+ } from "./messages";
57
72
  import {
58
73
  buildOpenAiNativeHistory,
59
74
  getPreservedOpenAiRemoteCompactionData,
@@ -224,7 +239,11 @@ export const DEFAULT_COMPACTION_SETTINGS: CompactionSettings = {
224
239
  v2RetainedMessageBudget: V2_RETAINED_MESSAGE_TOKEN_BUDGET,
225
240
  };
226
241
 
227
- /** Whether a compaction candidate preserves provider-native transport under the effective settings. */
242
+ /**
243
+ * Whether a compaction candidate preserves provider-native transport under the
244
+ * effective settings: an OpenAI Responses compact route (V1 or streamed V2) or
245
+ * the Anthropic compaction beta.
246
+ */
228
247
  export function shouldUseProviderNativeCompaction(
229
248
  model: Model,
230
249
  settings: Pick<CompactionSettings, "remoteEnabled" | "remoteStreamingV2Enabled">,
@@ -232,7 +251,8 @@ export function shouldUseProviderNativeCompaction(
232
251
  if (settings.remoteEnabled === false) return false;
233
252
  return (
234
253
  shouldUseOpenAiRemoteCompaction(model) ||
235
- (settings.remoteStreamingV2Enabled !== false && shouldUseCompactionV2Streaming(model))
254
+ (settings.remoteStreamingV2Enabled !== false && shouldUseCompactionV2Streaming(model)) ||
255
+ shouldUseAnthropicNativeCompaction(model)
236
256
  );
237
257
  }
238
258
 
@@ -395,22 +415,6 @@ export function resolveThresholdTokens(contextWindow: number, settings: Compacti
395
415
  // Cut point detection
396
416
  // ============================================================================
397
417
 
398
- function estimateEntriesTokens(
399
- entries: SessionEntry[],
400
- tokenizer: Tokenizer,
401
- startIndex: number,
402
- endIndex: number,
403
- ): number {
404
- let total = 0;
405
- for (let i = startIndex; i < endIndex; i++) {
406
- const msg = getMessageFromEntry(entries[i]);
407
- if (msg) {
408
- total += tokenizer.countMessage(msg);
409
- }
410
- }
411
- return total;
412
- }
413
-
414
418
  /**
415
419
  * Find valid cut points: indices of user, assistant, custom, or bashExecution messages.
416
420
  * Never cut at tool results (they must follow their tool call).
@@ -1252,6 +1256,8 @@ export interface CompactionPreparation {
1252
1256
  tokensBefore: number;
1253
1257
  /** Summary from previous compaction, for iterative update */
1254
1258
  previousSummary?: string;
1259
+ /** ISO timestamp of the previous compaction entry, for iterative update */
1260
+ previousSummaryTimestamp?: string;
1255
1261
  /** Preserved opaque compaction payload from the previous compaction, if any. */
1256
1262
  previousPreserveData?: Record<string, unknown>;
1257
1263
  /** File operations extracted from messagesToSummarize */
@@ -1331,16 +1337,10 @@ export function prepareCompaction(
1331
1337
 
1332
1338
  let prevCompactionIndex = findReadableCompactionIndex(pathEntries, settings, activeModel);
1333
1339
 
1334
- // Honor the latest `/clear` reset boundary. `/clear` records a
1335
- // `reset_boundary` marker and reports the model context empty, so compaction
1336
- // must not resurrect the dropped pre-clear turns into its summary — matching
1337
- // how buildSessionContext starts the model-context rebuild after the boundary.
1338
- // A boundary after the last reusable compaction supersedes it: the pre-reset
1339
- // summary was cleared too, so drop the previous-compaction reuse and start
1340
- // fresh after the boundary. A boundary at or before that compaction is already
1341
- // superseded by it, so only scan newer entries.
1340
+ // A reset after the reusable compaction clears its summary too. An older
1341
+ // reset still bounds how far we may recover that compaction's kept messages.
1342
1342
  let resetBoundaryIndex = -1;
1343
- for (let i = pathEntries.length - 1; i > prevCompactionIndex; i--) {
1343
+ for (let i = pathEntries.length - 1; i >= 0; i--) {
1344
1344
  if (pathEntries[i].type === "reset_boundary") {
1345
1345
  resetBoundaryIndex = i;
1346
1346
  break;
@@ -1349,14 +1349,43 @@ export function prepareCompaction(
1349
1349
  if (resetBoundaryIndex > prevCompactionIndex) {
1350
1350
  prevCompactionIndex = -1;
1351
1351
  }
1352
- const boundaryStart = Math.max(prevCompactionIndex, resetBoundaryIndex) + 1;
1353
- const boundaryEnd = pathEntries.length;
1352
+ const previousCompaction =
1353
+ prevCompactionIndex >= 0 ? (pathEntries[prevCompactionIndex] as CompactionEntry) : undefined;
1354
+ let boundaryStart = Math.max(prevCompactionIndex, resetBoundaryIndex) + 1;
1355
+ if (
1356
+ previousCompaction &&
1357
+ !getCompactionV2PreserveData(previousCompaction.preserveData) &&
1358
+ !getPreservedOpenAiRemoteCompactionData(previousCompaction.preserveData)
1359
+ ) {
1360
+ // Local summaries exclude the retained tail, whose original entries precede
1361
+ // the compaction record. Native replay already carries that tail. Only look
1362
+ // backwards: advisor snapshots put all retained messages after the summary
1363
+ // and may carry a keep ID from their previous, differently indexed snapshot.
1364
+ for (let i = resetBoundaryIndex + 1; i < prevCompactionIndex; i++) {
1365
+ if (pathEntries[i].id === previousCompaction.firstKeptEntryId) {
1366
+ boundaryStart = i;
1367
+ break;
1368
+ }
1369
+ }
1370
+ }
1371
+
1372
+ // Keep original IDs beside the converted messages so estimation, cutting,
1373
+ // and all three output regions share one sequence without journal metadata.
1374
+ const compactionEntries: SessionEntry[] = [];
1375
+ const compactionMessages: AgentMessage[] = [];
1376
+ for (let i = boundaryStart; i < pathEntries.length; i++) {
1377
+ const entry = pathEntries[i];
1378
+ const message = getMessageFromEntry(entry);
1379
+ if (!message) continue;
1380
+ compactionEntries.push(entry);
1381
+ compactionMessages.push(message);
1382
+ }
1354
1383
 
1355
1384
  const lastUsage = getLastAssistantUsage(pathEntries);
1356
1385
  const tokensBefore = lastUsage ? calculateContextTokens(lastUsage) : 0;
1357
1386
  let keepRecentTokens = settings.keepRecentTokens;
1358
1387
  if (lastUsage) {
1359
- const estimatedTokens = estimateEntriesTokens(pathEntries, tokenizer, boundaryStart, boundaryEnd);
1388
+ const estimatedTokens = tokenizer.countMessages(compactionMessages);
1360
1389
  const promptTokens = calculatePromptTokens(lastUsage);
1361
1390
  const ratio = estimatedTokens > 0 ? promptTokens / estimatedTokens : 0;
1362
1391
  if (Number.isFinite(ratio) && ratio > 1) {
@@ -1364,10 +1393,10 @@ export function prepareCompaction(
1364
1393
  }
1365
1394
  }
1366
1395
 
1367
- const cutPoint = findCutPoint(pathEntries, tokenizer, boundaryStart, boundaryEnd, keepRecentTokens);
1396
+ const cutPoint = findCutPoint(compactionEntries, tokenizer, 0, compactionEntries.length, keepRecentTokens);
1368
1397
 
1369
1398
  // Get ID of first kept entry
1370
- const firstKeptEntry = pathEntries[cutPoint.firstKeptEntryIndex];
1399
+ const firstKeptEntry = compactionEntries[cutPoint.firstKeptEntryIndex];
1371
1400
  if (!firstKeptEntry?.id) {
1372
1401
  return undefined; // Session needs migration
1373
1402
  }
@@ -1375,42 +1404,16 @@ export function prepareCompaction(
1375
1404
 
1376
1405
  const historyEnd = cutPoint.isSplitTurn ? cutPoint.turnStartIndex : cutPoint.firstKeptEntryIndex;
1377
1406
 
1378
- // Messages to summarize (will be discarded after summary)
1379
- const messagesToSummarize: AgentMessage[] = [];
1380
- for (let i = boundaryStart; i < historyEnd; i++) {
1381
- const msg = getMessageFromEntry(pathEntries[i]);
1382
- if (msg) messagesToSummarize.push(msg);
1383
- }
1384
-
1385
- // Messages for turn prefix summary (if splitting a turn)
1386
- const turnPrefixMessages: AgentMessage[] = [];
1387
- if (cutPoint.isSplitTurn) {
1388
- for (let i = cutPoint.turnStartIndex; i < cutPoint.firstKeptEntryIndex; i++) {
1389
- const msg = getMessageFromEntry(pathEntries[i]);
1390
- if (msg) turnPrefixMessages.push(msg);
1391
- }
1392
- }
1393
-
1394
- // Messages kept after compaction (recent history)
1395
- const recentMessages: AgentMessage[] = [];
1396
- for (let i = cutPoint.firstKeptEntryIndex; i < boundaryEnd; i++) {
1397
- const msg = getMessageFromEntry(pathEntries[i]);
1398
- if (msg) recentMessages.push(msg);
1399
- }
1407
+ const messagesToSummarize = compactionMessages.slice(0, historyEnd);
1408
+ const turnPrefixMessages = cutPoint.isSplitTurn
1409
+ ? compactionMessages.slice(cutPoint.turnStartIndex, cutPoint.firstKeptEntryIndex)
1410
+ : [];
1411
+ const recentMessages = compactionMessages.slice(cutPoint.firstKeptEntryIndex);
1400
1412
  // Nothing to summarize means compaction would be a no-op.
1401
1413
  if (messagesToSummarize.length === 0 && turnPrefixMessages.length === 0) {
1402
1414
  return undefined;
1403
1415
  }
1404
1416
 
1405
- // Get previous summary and preserved data for iterative updates
1406
- let previousSummary: string | undefined;
1407
- let previousPreserveData: Record<string, unknown> | undefined;
1408
- if (prevCompactionIndex >= 0) {
1409
- const prevCompaction = pathEntries[prevCompactionIndex] as CompactionEntry;
1410
- previousSummary = prevCompaction.summary;
1411
- previousPreserveData = prevCompaction.preserveData;
1412
- }
1413
-
1414
1417
  // Extract file operations from messages and previous compaction
1415
1418
  const fileOps = extractFileOperations(messagesToSummarize, pathEntries, prevCompactionIndex);
1416
1419
 
@@ -1428,8 +1431,9 @@ export function prepareCompaction(
1428
1431
  recentMessages,
1429
1432
  isSplitTurn: cutPoint.isSplitTurn,
1430
1433
  tokensBefore,
1431
- previousSummary,
1432
- previousPreserveData,
1434
+ previousSummary: previousCompaction?.summary,
1435
+ previousSummaryTimestamp: previousCompaction?.timestamp,
1436
+ previousPreserveData: previousCompaction?.preserveData,
1433
1437
  fileOps,
1434
1438
  settings,
1435
1439
  };
@@ -1585,6 +1589,9 @@ export async function compact(
1585
1589
  tools: options?.tools,
1586
1590
  fetch: options?.fetch,
1587
1591
  completeImpl: options?.completeImpl,
1592
+ // The caller's opt-out must reach every summarization oneshot, otherwise
1593
+ // an outer retry loop multiplies with the inner one (see SummaryOptions).
1594
+ oneshotRetry: options?.oneshotRetry,
1588
1595
  };
1589
1596
 
1590
1597
  const previousSnapcompactArchive = snapcompact.getPreservedArchive(previousPreserveData);
@@ -1599,7 +1606,10 @@ export async function compact(
1599
1606
  ? createSnapcompactArchiveMigrationMessage(previousSnapcompactArchiveText)
1600
1607
  : undefined;
1601
1608
 
1602
- let preserveData = withOpenAiRemoteCompactionPreserveData(previousPreserveData, undefined);
1609
+ let preserveData = withAnthropicCompactionPreserveData(
1610
+ withOpenAiRemoteCompactionPreserveData(previousPreserveData, undefined),
1611
+ undefined,
1612
+ );
1603
1613
  const remoteMessages: AgentMessage[] = [
1604
1614
  ...(snapcompactArchiveMigrationMessage ? [snapcompactArchiveMigrationMessage] : []),
1605
1615
  ...messagesToSummarize,
@@ -1784,6 +1794,112 @@ export async function compact(
1784
1794
  }
1785
1795
  }
1786
1796
 
1797
+ // Anthropic server-side compaction: the live turn's request shape plus the
1798
+ // compact edit, so the API summarizes from its cached prefix. The summary is
1799
+ // real text, persisted both as the entry summary and as the native replay
1800
+ // payload. A context below the API's trigger floor cannot compact remotely
1801
+ // and takes the local summarizer instead — an eligibility boundary, not a
1802
+ // failure.
1803
+ let nativeSummary: string | undefined;
1804
+ let nativeEncryptedContent: string | undefined;
1805
+ let nativeUsedTokens: number | undefined;
1806
+ if (
1807
+ !usedRemoteCompaction &&
1808
+ settings.remoteEnabled !== false &&
1809
+ shouldUseAnthropicNativeCompaction(model) &&
1810
+ tokensBefore >= ANTHROPIC_COMPACTION_MIN_CONTEXT_TOKENS
1811
+ ) {
1812
+ const previousNative = getPreservedAnthropicCompactionData(previousPreserveData);
1813
+ // Lead with the previous summary exactly as the live context renders it:
1814
+ // natively when this provider wrote it, as text otherwise. The request
1815
+ // then shares the live turn's prefix byte-for-byte. A prior snapcompact
1816
+ // archive is already merged into that summary text, so the archive
1817
+ // migration message the OpenAI lanes carry is omitted here.
1818
+ // The rewrite marker must precede every message this request replays —
1819
+ // summarized history and retained tail alike — exactly like the live
1820
+ // context rebuild predates its tail. A previous compaction's commit
1821
+ // timestamp is newer than re-retained or re-summarized turns, so
1822
+ // reusing it would strip their bound thinking only in this request,
1823
+ // diverging from the cached live prefix (and possibly dropping below
1824
+ // the trigger). Manually built preparations with no input at all fall
1825
+ // back to the current time.
1826
+ const firstReplayed = messagesToSummarize[0] ?? turnPrefixMessages[0] ?? recentMessages[0];
1827
+ const previousSummaryAt =
1828
+ firstReplayed !== undefined ? new Date(firstReplayed.timestamp - 1).toISOString() : new Date().toISOString();
1829
+ const previousSummaryMessage = previousSummaryForCompaction
1830
+ ? createCompactionSummaryMessage(previousSummaryForCompaction, tokensBefore, previousSummaryAt, {
1831
+ providerPayload:
1832
+ previousNative?.provider === model.provider
1833
+ ? {
1834
+ type: "anthropicCompaction",
1835
+ provider: previousNative.provider,
1836
+ content: previousNative.content,
1837
+ ...(previousNative.encryptedContent
1838
+ ? { encryptedContent: previousNative.encryptedContent }
1839
+ : {}),
1840
+ ...(previousNative.filesText ? { filesText: previousNative.filesText } : {}),
1841
+ }
1842
+ : undefined,
1843
+ })
1844
+ : undefined;
1845
+ const convertToLlm = summaryOptions.convertToLlm ?? defaultConvertToLlm;
1846
+ const retainedTail = convertToLlm(recentMessages);
1847
+ const messages = [
1848
+ ...convertToLlm([
1849
+ ...(previousSummaryMessage ? [previousSummaryMessage] : []),
1850
+ ...messagesToSummarize,
1851
+ ...turnPrefixMessages,
1852
+ ]),
1853
+ ...retainedTail,
1854
+ ];
1855
+ try {
1856
+ const remote = await requestAnthropicNativeCompaction(
1857
+ model,
1858
+ apiKey,
1859
+ {
1860
+ systemPrompt: summaryOptions.remoteSystemPrompt ?? [SUMMARIZATION_SYSTEM_PROMPT],
1861
+ messages,
1862
+ tools: summaryOptions.tools,
1863
+ instructions: buildAnthropicCompactionInstructions(
1864
+ summaryOptions.promptOverride ?? SUMMARIZATION_PROMPT,
1865
+ customInstructions,
1866
+ formatAdditionalContext(summaryOptions.extraContext).trim() || undefined,
1867
+ describeRetainedTail(retainedTail),
1868
+ ),
1869
+ maxTokens: Math.min(Math.floor(0.8 * reserveTokens), MAX_SUMMARY_TOKENS),
1870
+ reasoning: resolveCompactionEffort(model, summaryOptions.thinkingLevel),
1871
+ },
1872
+ signal,
1873
+ {
1874
+ initiatorOverride: summaryOptions.initiatorOverride,
1875
+ metadata: summaryOptions.metadata,
1876
+ fetch: summaryOptions.fetch,
1877
+ sessionId: summaryOptions.sessionId,
1878
+ promptCacheKey: summaryOptions.promptCacheKey,
1879
+ providerSessionState: summaryOptions.providerSessionState,
1880
+ completeImpl: summaryOptions.completeImpl,
1881
+ telemetry: summaryOptions.telemetry,
1882
+ retry: summaryOneshotRetry(summaryOptions),
1883
+ },
1884
+ );
1885
+ nativeSummary = remote.content;
1886
+ nativeEncryptedContent = remote.encryptedContent;
1887
+ nativeUsedTokens = calculatePromptTokens(remote.usage);
1888
+ usedRemoteCompaction = true;
1889
+ } catch (err) {
1890
+ // A user/session abort is a cancellation, not a remote failure —
1891
+ // swallowing it here would downgrade Esc into "fall back to local
1892
+ // summarization" and keep compaction running on an aborted signal.
1893
+ if (signal?.aborted) throw err;
1894
+ nativeCompactionError = selectNativeCompactionError(nativeCompactionError, err);
1895
+ logger.warn("Anthropic server-side compaction failed", {
1896
+ error: err instanceof Error ? err.message : String(err),
1897
+ model: model.id,
1898
+ provider: model.provider,
1899
+ });
1900
+ }
1901
+ }
1902
+
1787
1903
  if (!usedRemoteCompaction && nativeCompactionError !== undefined && !summaryOptions.remoteEndpoint) {
1788
1904
  throw new NativeCompactionError(nativeCompactionError);
1789
1905
  }
@@ -1791,7 +1907,11 @@ export async function compact(
1791
1907
  // Generate summaries (can be parallel if both needed) and merge into one
1792
1908
  let summary: string;
1793
1909
 
1794
- if (usedRemoteCompaction) {
1910
+ if (nativeSummary !== undefined) {
1911
+ // The API wrote a real summary; it is the entry text. The replayed
1912
+ // block below stays verbatim so it matches the opaque state.
1913
+ summary = nativeSummary;
1914
+ } else if (usedRemoteCompaction) {
1795
1915
  // Remote compaction (V2 or V1) already compacted remotely; the durable
1796
1916
  // history lives in the provider replay payload (preserveData). Skip local
1797
1917
  // summarization so a successful remote compaction never pays for a second,
@@ -1850,6 +1970,22 @@ export async function compact(
1850
1970
  // Compute file lists and append to summary
1851
1971
  const { readFiles, modifiedFiles } = computeFileLists(fileOps);
1852
1972
  summary = upsertFileOperations(summary, readFiles, modifiedFiles, fileOps.read);
1973
+ if (nativeSummary !== undefined) {
1974
+ // The replayed block stays byte-identical to the API's summary so it
1975
+ // matches `encryptedContent`. The harness file lists above travel
1976
+ // separately: the converter replaces the summary message with the
1977
+ // block and skips its text, so they would otherwise be invisible to
1978
+ // this provider. Every other provider keeps reading the entry text.
1979
+ const filesText = upsertFileOperations("", readFiles, modifiedFiles, fileOps.read) || undefined;
1980
+ preserveData = withAnthropicCompactionPreserveData(preserveData, {
1981
+ provider: model.provider,
1982
+ content: nativeSummary,
1983
+ ...(nativeEncryptedContent ? { encryptedContent: nativeEncryptedContent } : {}),
1984
+ ...(filesText ? { filesText } : {}),
1985
+ model: model.id,
1986
+ usedTokens: nativeUsedTokens,
1987
+ });
1988
+ }
1853
1989
 
1854
1990
  if (!firstKeptEntryId) {
1855
1991
  throw new Error("First kept entry has no ID - session may need migration");
@@ -2,6 +2,7 @@
2
2
  * Compaction and summarization utilities.
3
3
  */
4
4
 
5
+ export * from "./anthropic";
5
6
  export * from "./branch-summarization";
6
7
  export * from "./compaction";
7
8
  export * from "./entries";
@@ -66,14 +66,14 @@ export * from "./compaction-v2-streaming";
66
66
  export const OPENAI_REMOTE_COMPACTION_PRESERVE_KEY = "openaiRemoteCompaction";
67
67
 
68
68
  /**
69
- * Hard ceiling on remote compaction HTTP requests. Unlike every provider
70
- * stream (guarded by first-event/idle watchdogs in pi-ai), these are raw
71
- * fetches awaiting one non-streamed JSON body — a connection silently dropped
72
- * by a middlebox would otherwise hang the whole compaction pipeline forever
73
- * (frozen "Auto context-full maintenance…" spinner, manual /compact queueing
74
- * behind it). On timeout the caller falls back to local summarization.
69
+ * Hard ceiling on remote compaction HTTP requests (5 minutes). Unlike every
70
+ * provider stream (guarded by first-event/idle watchdogs in pi-ai), these are
71
+ * raw fetches awaiting one non-streamed JSON body — a connection silently
72
+ * dropped by a middlebox would otherwise hang the whole compaction pipeline
73
+ * forever (frozen "Auto context-full maintenance…" spinner, manual /compact
74
+ * queueing behind it). On timeout the caller falls back to local summarization.
75
75
  */
76
- export const REMOTE_COMPACTION_TIMEOUT_MS = 180_000;
76
+ export const REMOTE_COMPACTION_TIMEOUT_MS = 300_000;
77
77
 
78
78
  const DEFAULT_AZURE_API_VERSION = "v1";
79
79
 
@@ -0,0 +1,17 @@
1
+ {{#if retainedTail}}
2
+ SCOPE: The conversation's final {{#when retainedTail.count "==" 1}}{{retainedTail.role}} message stays{{else}}{{retainedTail.count}} messages, starting with a {{retainedTail.role}} message, stay{{/when}} in context verbatim after your summary. Summarize ONLY the history before those messages. You MUST NOT restate anything from those final messages — the reader sees them right after the summary — and you MUST treat them as the most recent state when describing progress and next steps.
3
+ {{else}}
4
+ SCOPE: The conversation above is the transcript to summarize. The API replaces everything before your summary with it, so nothing you leave out survives into the next context window.
5
+ {{/if}}
6
+
7
+ {{#if extraContext}}
8
+ {{extraContext}}
9
+
10
+ {{/if}}
11
+ {{basePrompt}}
12
+ {{#if customInstructions}}
13
+
14
+ Additional focus: {{customInstructions}}
15
+ {{/if}}
16
+
17
+ You MUST NOT call any tools while writing the summary; respond with the summary text only.
package/src/proxy.ts CHANGED
@@ -20,7 +20,6 @@ import {
20
20
  type StreamingPartialJsonCarrier,
21
21
  setStreamingPartialJson,
22
22
  } from "@oh-my-pi/pi-ai/utils/block-symbols";
23
- import { calculateCost } from "@oh-my-pi/pi-catalog/models";
24
23
  import { parseStreamingJson, readSseJson } from "@oh-my-pi/pi-utils";
25
24
 
26
25
  // Event stream adapter for proxy SSE events
@@ -171,7 +170,7 @@ export function streamProxy(model: Model, context: Context, options: ProxyStream
171
170
  response.body as ReadableStream<Uint8Array>,
172
171
  options.signal,
173
172
  )) {
174
- const parsedEvent = processProxyEvent(model, event, partial, partialJsonByIndex);
173
+ const parsedEvent = processProxyEvent(event, partial, partialJsonByIndex);
175
174
  if (parsedEvent) {
176
175
  if (parsedEvent.type === "done" || parsedEvent.type === "error") {
177
176
  sawTerminalEvent = true;
@@ -233,7 +232,6 @@ function scrubPartialJson(partial: AssistantMessage): void {
233
232
  * reads as still-streaming.
234
233
  */
235
234
  function processProxyEvent(
236
- model: Model,
237
235
  proxyEvent: ProxyAssistantMessageEvent,
238
236
  partial: AssistantMessage,
239
237
  partialJsonByIndex: Map<number, string>,
@@ -375,7 +373,6 @@ function processProxyEvent(
375
373
  partial.stopReason = proxyEvent.reason;
376
374
  partial.usage = proxyEvent.usage;
377
375
  if (proxyEvent.content !== undefined) partial.content = proxyEvent.content;
378
- calculateCost(model, partial.usage);
379
376
  scrubPartialJson(partial);
380
377
  return { type: "done", reason: proxyEvent.reason, message: partial };
381
378
 
@@ -384,7 +381,6 @@ function processProxyEvent(
384
381
  partial.errorMessage = proxyEvent.errorMessage;
385
382
  partial.usage = proxyEvent.usage;
386
383
  if (proxyEvent.content !== undefined) partial.content = proxyEvent.content;
387
- calculateCost(model, partial.usage);
388
384
  scrubPartialJson(partial);
389
385
  return { type: "error", reason: proxyEvent.reason, error: partial };
390
386
  }
package/src/tokenizer.ts CHANGED
@@ -219,14 +219,22 @@ export class Tokenizer {
219
219
  }
220
220
 
221
221
  switch (message.role) {
222
- case "user": {
223
- const content: string | Array<{ type: string; text?: string }> = message.content;
222
+ case "user":
223
+ case "developer": {
224
+ // Both roles carry `string | (TextContent | ImageContent)[]` and both are
225
+ // sent to the provider -- convertMessageToLlm handles developer alongside
226
+ // user -- so they are counted alike. Without the developer case the switch
227
+ // fell through to `default: return 0`, and the old annotation narrowed the
228
+ // blocks to text-only, hiding the image charge the toolResult arm applies.
229
+ const content = message.content;
224
230
  if (typeof content === "string") {
225
231
  fragments.push(content);
226
232
  } else if (Array.isArray(content)) {
227
233
  for (const block of content) {
228
234
  if (block.type === "text" && block.text) {
229
235
  fragments.push(block.text);
236
+ } else if (block.type === "image") {
237
+ extra += IMAGE_TOKEN_ESTIMATE;
230
238
  }
231
239
  }
232
240
  }