@oh-my-pi/pi-agent-core 18.1.16 → 18.1.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +23 -0
- package/dist/types/compaction/anthropic.d.ts +108 -0
- package/dist/types/compaction/compaction-v2-streaming.d.ts +2 -2
- package/dist/types/compaction/compaction.d.ts +7 -1
- package/dist/types/compaction/index.d.ts +1 -0
- package/dist/types/compaction/openai.d.ts +7 -7
- package/package.json +8 -8
- package/src/agent-loop.ts +64 -2
- package/src/compaction/anthropic.ts +310 -0
- package/src/compaction/compaction-v2-streaming.ts +2 -2
- package/src/compaction/compaction.ts +204 -68
- package/src/compaction/index.ts +1 -0
- package/src/compaction/openai.ts +7 -7
- package/src/compaction/prompts/anthropic-compaction-instructions.md +17 -0
- package/src/proxy.ts +1 -5
- package/src/tokenizer.ts +10 -2
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,29 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [18.1.18] - 2026-09-11
|
|
6
|
+
|
|
7
|
+
### Added
|
|
8
|
+
|
|
9
|
+
- Anthropic server-side compaction as a `remote` compaction backend: model lines the beta supports (`compat.supportsServerCompaction`, rule-owned in the catalog: Opus 4.6+, Sonnet 4.6+, Fable/Mythos 5) on the official endpoint, resolved the way the provider routes requests, plus Anthropic-compatible routes with `remoteCompaction.enabled`, compact by re-issuing the live turn's own request — same system prompt, tools, and history, so it reads the prompt cache the last turn wrote — with the `compact_20260112` edit paused after the summary and the harness summary prompt as `instructions`. The instructions name where the retained tail begins so the summary covers only the history the rebuilt context drops. The API's summary is stored as the entry text and as `preserveData.anthropicCompaction`, replayed natively on later Anthropic requests and read as plain text by every other provider; the retained tail comes from session entries as with a local summary. Contexts below 55k tokens (the API trigger floor plus margin) keep summarizing locally, and a response without a summary is a native failure, like the OpenAI lanes. An aborted compaction response is the abort (a cancellation, never a native failure) and an error response keeps its HTTP status, so auth and timeout classification match the OpenAI lanes; the block's opaque `encrypted_content` is persisted as `preserveData.anthropicCompaction.encryptedContent` and replayed verbatim.
|
|
10
|
+
|
|
11
|
+
### Fixed
|
|
12
|
+
|
|
13
|
+
- `compact()` now forwards the caller's `oneshotRetry` opt-out to every summarization oneshot; auto-compaction's outer retry loop no longer multiplies with the inner transient-failure retries.
|
|
14
|
+
|
|
15
|
+
## [18.1.17] - 2026-09-10
|
|
16
|
+
|
|
17
|
+
### Changed
|
|
18
|
+
|
|
19
|
+
- `Tool <name> not found` now names a plausible intended target when the advertised set contains one, e.g. `Tool mcp__abc123__xyz789_read not found. Did you mean read?`. A model that mis-transcribes a long opaque tool name reliably keeps the trailing segment, which is the only part carrying meaning, so the miss becomes recoverable in the same turn instead of costing a round trip. Purely advisory — the suggestion is only ever a string in the error, never a dispatch target, so an unrecognized name still fails ([#10109](https://github.com/can1357/oh-my-pi/issues/10109) by [@oldschoola](https://github.com/oldschoola)).
|
|
20
|
+
|
|
21
|
+
### Fixed
|
|
22
|
+
|
|
23
|
+
- Fixed the token estimator counting developer messages as free and ignoring images in user content, which let context budgeting, pruning and the compaction trigger read a transcript as far smaller than the one sent to the provider.
|
|
24
|
+
- Fixed repeated local compaction omitting messages retained before the previous compaction record, while preserving original entry IDs and `/clear` boundaries.
|
|
25
|
+
- Raised remote compaction request timeout from 3 minutes to 5 minutes so long Codex/gpt-6-astra compact streams can finish before the watchdog aborts them.
|
|
26
|
+
- Fixed proxy responses dropping the cost the server reported; recorded costs are kept instead of being recomputed.
|
|
27
|
+
|
|
5
28
|
## [18.1.10] - 2026-09-04
|
|
6
29
|
|
|
7
30
|
### Fixed
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Anthropic server-side compaction (`compact-2026-01-12` beta).
|
|
3
|
+
*
|
|
4
|
+
* The compaction request is the live turn's own request shape — same system
|
|
5
|
+
* prompt, tools, and message history — plus the `compact_20260112` edit with
|
|
6
|
+
* `pause_after_compaction`. The API summarizes the prompt from the already
|
|
7
|
+
* cached prefix and stops; the summary arrives as a `compaction` block that
|
|
8
|
+
* the provider surfaces as an `anthropicCompaction` payload. The summary is
|
|
9
|
+
* plain text, so it doubles as the compaction entry's readable summary for
|
|
10
|
+
* every other provider, while the Anthropic provider replays it as a native
|
|
11
|
+
* block (the API drops everything that precedes it). The retained tail after
|
|
12
|
+
* the cut point is replayed from session entries exactly like a local summary.
|
|
13
|
+
*/
|
|
14
|
+
import type { AnthropicCompactionPayload, ApiKey, Effort, Message, Model, SimpleStreamOptions, Tool, Usage } from "@oh-my-pi/pi-ai";
|
|
15
|
+
import { type InstrumentedChatSpanOptions } from "../telemetry.js";
|
|
16
|
+
export declare const ANTHROPIC_COMPACTION_PRESERVE_KEY = "anthropicCompaction";
|
|
17
|
+
/** The API rejects a `compact_20260112` trigger below this many input tokens. */
|
|
18
|
+
export declare const ANTHROPIC_COMPACTION_MIN_TRIGGER_TOKENS = 50000;
|
|
19
|
+
/**
|
|
20
|
+
* Smallest context the native lane accepts. The trigger sits at the API
|
|
21
|
+
* floor, so a prompt that lands below it is answered instead of compacted;
|
|
22
|
+
* the margin over the floor absorbs the difference between the last reported
|
|
23
|
+
* context size and the compaction request's own input.
|
|
24
|
+
*/
|
|
25
|
+
export declare const ANTHROPIC_COMPACTION_MIN_CONTEXT_TOKENS = 55000;
|
|
26
|
+
/** Summary persisted under {@link ANTHROPIC_COMPACTION_PRESERVE_KEY}. */
|
|
27
|
+
export interface AnthropicCompactionPreserveData {
|
|
28
|
+
provider: string;
|
|
29
|
+
content: string;
|
|
30
|
+
/** Opaque provider state the API attached to the block; replayed verbatim. */
|
|
31
|
+
encryptedContent?: string;
|
|
32
|
+
/** Harness file metadata (`<files>` section) replayed after the native block. */
|
|
33
|
+
filesText?: string;
|
|
34
|
+
/** Model that wrote the summary. */
|
|
35
|
+
model?: string;
|
|
36
|
+
/** Prompt tokens the compaction request processed, for display. */
|
|
37
|
+
usedTokens?: number;
|
|
38
|
+
}
|
|
39
|
+
/**
|
|
40
|
+
* Whether a model compacts through the Anthropic compaction beta. Model
|
|
41
|
+
* eligibility is catalog policy (`compat.supportsServerCompaction`, the
|
|
42
|
+
* lineage the beta documents); endpoint eligibility is resolved the way the
|
|
43
|
+
* provider routes requests, so a Foundry or `ANTHROPIC_BASE_URL` reroute of a
|
|
44
|
+
* first-party model is excluded unless the route opted in with
|
|
45
|
+
* `remoteCompaction.enabled`.
|
|
46
|
+
*/
|
|
47
|
+
export declare function shouldUseAnthropicNativeCompaction(model: Model): model is Model<"anthropic-messages">;
|
|
48
|
+
export declare function getPreservedAnthropicCompactionData(preserveData: Record<string, unknown> | undefined): AnthropicCompactionPreserveData | undefined;
|
|
49
|
+
/** Set or strip the Anthropic compaction slot; a new compaction never inherits a stale summary. */
|
|
50
|
+
export declare function withAnthropicCompactionPreserveData(preserveData: Record<string, unknown> | undefined, compaction: AnthropicCompactionPreserveData | undefined): Record<string, unknown> | undefined;
|
|
51
|
+
/** Replay payload for a compaction summary the active model produced natively. */
|
|
52
|
+
export declare function getAnthropicCompactionPayload(preserveData: Record<string, unknown> | undefined): AnthropicCompactionPayload | undefined;
|
|
53
|
+
/**
|
|
54
|
+
* The retained tail as the model will see it, for the summarization
|
|
55
|
+
* instructions: how many of the conversation's final wire messages stay in
|
|
56
|
+
* context verbatim, and the role of the first. The compaction request carries
|
|
57
|
+
* the whole conversation so the prompt cache the live turn wrote is read, but
|
|
58
|
+
* the summary must cover only the history before that tail — the local
|
|
59
|
+
* summarizer never sees the tail, and the rebuilt context replays it after the
|
|
60
|
+
* summary. Counting mirrors the provider's message conversion (consecutive
|
|
61
|
+
* tool results collapse into one user message; developer messages are user
|
|
62
|
+
* messages). Structured for the prompt template, which renders the
|
|
63
|
+
* singular/plural wording; the description quotes no content: quoting the
|
|
64
|
+
* tail would hand the summarizer the very facts it must leave to the tail.
|
|
65
|
+
*/
|
|
66
|
+
export interface RetainedTailScope {
|
|
67
|
+
count: number;
|
|
68
|
+
role: "assistant" | "user";
|
|
69
|
+
}
|
|
70
|
+
export declare function describeRetainedTail(messages: readonly Message[]): RetainedTailScope | undefined;
|
|
71
|
+
/**
|
|
72
|
+
* Summarization prompt sent as the edit's `instructions`, which replace the
|
|
73
|
+
* API default entirely. The template lays out the retained-tail boundary
|
|
74
|
+
* first, so the summary covers only the history the rebuilt context drops,
|
|
75
|
+
* then the caller's extra context, the same structure prompt as the local
|
|
76
|
+
* summarizer, the caller's focus, and the tool-abstention clause the API
|
|
77
|
+
* recommends when tools are defined (a summarization pass that calls a tool
|
|
78
|
+
* yields no summary).
|
|
79
|
+
*/
|
|
80
|
+
export declare function buildAnthropicCompactionInstructions(basePrompt: string, customInstructions: string | undefined, extraContext: string | undefined, retainedTail: RetainedTailScope | undefined): string;
|
|
81
|
+
export interface AnthropicNativeCompactionRequest {
|
|
82
|
+
systemPrompt: string[];
|
|
83
|
+
messages: Message[];
|
|
84
|
+
tools?: Tool[];
|
|
85
|
+
instructions: string;
|
|
86
|
+
maxTokens: number;
|
|
87
|
+
reasoning?: Effort;
|
|
88
|
+
}
|
|
89
|
+
export interface AnthropicNativeCompactionResponse {
|
|
90
|
+
content: string;
|
|
91
|
+
encryptedContent?: string;
|
|
92
|
+
usage: Usage;
|
|
93
|
+
model: string;
|
|
94
|
+
}
|
|
95
|
+
export interface AnthropicNativeCompactionOptions extends Pick<SimpleStreamOptions, "initiatorOverride" | "metadata" | "fetch" | "sessionId" | "promptCacheKey" | "providerSessionState" | "maxInFlightRequests">, Pick<InstrumentedChatSpanOptions, "completeImpl" | "telemetry" | "retry"> {
|
|
96
|
+
}
|
|
97
|
+
/**
|
|
98
|
+
* Run one compaction request and return the summary the API wrote, with the
|
|
99
|
+
* opaque `encrypted_content` the API attached for the replay. `completeSimple`
|
|
100
|
+
* resolves terminal failures as messages, so their classification is restored
|
|
101
|
+
* here: an aborted response is an `AbortError` (a cancellation, never a native
|
|
102
|
+
* failure) and an error response keeps its HTTP status, so auth and timeout
|
|
103
|
+
* handling downstream classify it the same way as the OpenAI lanes. A response
|
|
104
|
+
* without a summary is a native failure — the API answers the prompt instead
|
|
105
|
+
* when its input never reached the trigger, and returns an empty block when
|
|
106
|
+
* the model called a tool during summarization.
|
|
107
|
+
*/
|
|
108
|
+
export declare function requestAnthropicNativeCompaction(model: Model<"anthropic-messages">, apiKey: ApiKey, request: AnthropicNativeCompactionRequest, signal: AbortSignal | undefined, options: AnthropicNativeCompactionOptions): Promise<AnthropicNativeCompactionResponse>;
|
|
@@ -12,8 +12,8 @@ import { type OpenAICodexCompactionBody } from "@oh-my-pi/pi-ai/providers/openai
|
|
|
12
12
|
export declare const V2_RETAINED_MESSAGE_TOKEN_BUDGET = 64000;
|
|
13
13
|
/** Max retries for V2 streaming compaction on transient stream errors. */
|
|
14
14
|
export declare const V2_COMPACTION_MAX_RETRIES = 2;
|
|
15
|
-
/** Timeout for V2 streaming compaction (
|
|
16
|
-
export declare const V2_COMPACTION_TIMEOUT_MS =
|
|
15
|
+
/** Timeout for V2 streaming compaction (5 minutes, same as V1). */
|
|
16
|
+
export declare const V2_COMPACTION_TIMEOUT_MS = 300000;
|
|
17
17
|
/** Token usage reported by the streamed V2 Responses completion. */
|
|
18
18
|
export interface CompactionV2Usage {
|
|
19
19
|
inputTokens: number;
|
|
@@ -65,7 +65,11 @@ export declare const DEFAULT_RESERVE_TOKENS = 16384;
|
|
|
65
65
|
*/
|
|
66
66
|
export declare const MAX_SUMMARY_TOKENS = 16384;
|
|
67
67
|
export declare const DEFAULT_COMPACTION_SETTINGS: CompactionSettings;
|
|
68
|
-
/**
|
|
68
|
+
/**
|
|
69
|
+
* Whether a compaction candidate preserves provider-native transport under the
|
|
70
|
+
* effective settings: an OpenAI Responses compact route (V1 or streamed V2) or
|
|
71
|
+
* the Anthropic compaction beta.
|
|
72
|
+
*/
|
|
69
73
|
export declare function shouldUseProviderNativeCompaction(model: Model, settings: Pick<CompactionSettings, "remoteEnabled" | "remoteStreamingV2Enabled">): boolean;
|
|
70
74
|
/**
|
|
71
75
|
* Calculate total context tokens from usage.
|
|
@@ -290,6 +294,8 @@ export interface CompactionPreparation {
|
|
|
290
294
|
tokensBefore: number;
|
|
291
295
|
/** Summary from previous compaction, for iterative update */
|
|
292
296
|
previousSummary?: string;
|
|
297
|
+
/** ISO timestamp of the previous compaction entry, for iterative update */
|
|
298
|
+
previousSummaryTimestamp?: string;
|
|
293
299
|
/** Preserved opaque compaction payload from the previous compaction, if any. */
|
|
294
300
|
previousPreserveData?: Record<string, unknown>;
|
|
295
301
|
/** File operations extracted from messagesToSummarize */
|
|
@@ -19,14 +19,14 @@ import { Tokenizer } from "../tokenizer.js";
|
|
|
19
19
|
export * from "./compaction-v2-streaming.js";
|
|
20
20
|
export declare const OPENAI_REMOTE_COMPACTION_PRESERVE_KEY = "openaiRemoteCompaction";
|
|
21
21
|
/**
|
|
22
|
-
* Hard ceiling on remote compaction HTTP requests. Unlike every
|
|
23
|
-
* stream (guarded by first-event/idle watchdogs in pi-ai), these are
|
|
24
|
-
* fetches awaiting one non-streamed JSON body — a connection silently
|
|
25
|
-
* by a middlebox would otherwise hang the whole compaction pipeline
|
|
26
|
-
* (frozen "Auto context-full maintenance…" spinner, manual /compact
|
|
27
|
-
* behind it). On timeout the caller falls back to local summarization.
|
|
22
|
+
* Hard ceiling on remote compaction HTTP requests (5 minutes). Unlike every
|
|
23
|
+
* provider stream (guarded by first-event/idle watchdogs in pi-ai), these are
|
|
24
|
+
* raw fetches awaiting one non-streamed JSON body — a connection silently
|
|
25
|
+
* dropped by a middlebox would otherwise hang the whole compaction pipeline
|
|
26
|
+
* forever (frozen "Auto context-full maintenance…" spinner, manual /compact
|
|
27
|
+
* queueing behind it). On timeout the caller falls back to local summarization.
|
|
28
28
|
*/
|
|
29
|
-
export declare const REMOTE_COMPACTION_TIMEOUT_MS =
|
|
29
|
+
export declare const REMOTE_COMPACTION_TIMEOUT_MS = 300000;
|
|
30
30
|
export declare const CONTEXT_WINDOW_TRUNCATED_OUTPUT_MESSAGE: string;
|
|
31
31
|
export interface TrimRemoteCompactionInputResult {
|
|
32
32
|
input: Array<Record<string, unknown>>;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@oh-my-pi/pi-agent-core",
|
|
4
|
-
"version": "18.1.
|
|
4
|
+
"version": "18.1.18",
|
|
5
5
|
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
|
|
6
6
|
"homepage": "https://omp.sh",
|
|
7
7
|
"author": "Stencil Labs, Inc.",
|
|
@@ -35,16 +35,16 @@
|
|
|
35
35
|
"fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
|
|
36
36
|
},
|
|
37
37
|
"dependencies": {
|
|
38
|
-
"@oh-my-pi/pi-ai": "18.1.
|
|
39
|
-
"@oh-my-pi/pi-catalog": "18.1.
|
|
40
|
-
"@oh-my-pi/pi-natives": "18.1.
|
|
41
|
-
"@oh-my-pi/pi-utils": "18.1.
|
|
42
|
-
"@oh-my-pi/pi-wire": "18.1.
|
|
43
|
-
"@oh-my-pi/snapcompact": "18.1.
|
|
38
|
+
"@oh-my-pi/pi-ai": "18.1.18",
|
|
39
|
+
"@oh-my-pi/pi-catalog": "18.1.18",
|
|
40
|
+
"@oh-my-pi/pi-natives": "18.1.18",
|
|
41
|
+
"@oh-my-pi/pi-utils": "18.1.18",
|
|
42
|
+
"@oh-my-pi/pi-wire": "18.1.18",
|
|
43
|
+
"@oh-my-pi/snapcompact": "18.1.18",
|
|
44
44
|
"@opentelemetry/api": "^1.9.1"
|
|
45
45
|
},
|
|
46
46
|
"devDependencies": {
|
|
47
|
-
"@oh-my-pi/omptype": "18.1.
|
|
47
|
+
"@oh-my-pi/omptype": "18.1.18",
|
|
48
48
|
"@opentelemetry/context-async-hooks": "^2.9.0",
|
|
49
49
|
"@opentelemetry/sdk-trace-base": "^2.9.0",
|
|
50
50
|
"@types/bun": "^1.3.14"
|
package/src/agent-loop.ts
CHANGED
|
@@ -2269,6 +2269,68 @@ function resolveToolForCall(
|
|
|
2269
2269
|
);
|
|
2270
2270
|
}
|
|
2271
2271
|
|
|
2272
|
+
/** Shortest suggestable segment; below this the match is noise (`id`, `to`). */
|
|
2273
|
+
const MIN_TOOL_NAME_SUGGESTION_SEGMENT = 3;
|
|
2274
|
+
/** Cap on names listed for an ambiguous miss, so the error stays readable. */
|
|
2275
|
+
const MAX_TOOL_NAME_SUGGESTIONS = 3;
|
|
2276
|
+
|
|
2277
|
+
/**
|
|
2278
|
+
* Advertised tool names sharing a trailing `_`-delimited segment with `name`.
|
|
2279
|
+
*
|
|
2280
|
+
* A model that mis-transcribes a long opaque tool name reliably keeps the
|
|
2281
|
+
* trailing verb — that segment is the only part carrying meaning, while any
|
|
2282
|
+
* leading id segments are high-entropy and mnemonic-free. Matching on it turns
|
|
2283
|
+
* an otherwise dead `not found` into a self-correcting one.
|
|
2284
|
+
*
|
|
2285
|
+
* Both the last `__` and last `_` boundary are tried, so a name that lost only
|
|
2286
|
+
* its separator (`…__resolve_library_id`) and one that lost a whole id segment
|
|
2287
|
+
* (`…__read`) both recover. Purely advisory: this only builds an error string
|
|
2288
|
+
* and never selects a tool, so dispatch semantics are unchanged.
|
|
2289
|
+
*/
|
|
2290
|
+
function suggestToolNames(
|
|
2291
|
+
name: string,
|
|
2292
|
+
tools: ReadonlyArray<Pick<AgentTool, "name" | "customWireName">> | undefined,
|
|
2293
|
+
): string[] {
|
|
2294
|
+
if (!tools || tools.length === 0) return [];
|
|
2295
|
+
const segments: string[] = [];
|
|
2296
|
+
for (const boundary of ["__", "_"]) {
|
|
2297
|
+
const idx = name.lastIndexOf(boundary);
|
|
2298
|
+
if (idx < 0) continue;
|
|
2299
|
+
const segment = name.slice(idx + boundary.length);
|
|
2300
|
+
if (segment.length >= MIN_TOOL_NAME_SUGGESTION_SEGMENT && !segments.includes(segment)) segments.push(segment);
|
|
2301
|
+
}
|
|
2302
|
+
if (segments.length === 0) return [];
|
|
2303
|
+
// Longest tail first. A distinctive `__` tail (`resolve_library_get`) is a
|
|
2304
|
+
// far stronger signal than the generic `_` tail it contains (`get`), and the
|
|
2305
|
+
// caller truncates the list — so the strongest match has to sort ahead of
|
|
2306
|
+
// however many tools happen to share the weak one.
|
|
2307
|
+
segments.sort((a, b) => b.length - a.length);
|
|
2308
|
+
const matches: string[] = [];
|
|
2309
|
+
for (const segment of segments) {
|
|
2310
|
+
for (const tool of tools) {
|
|
2311
|
+
for (const candidate of [tool.name, tool.customWireName]) {
|
|
2312
|
+
if (candidate === undefined || candidate === name || matches.includes(candidate)) continue;
|
|
2313
|
+
if (candidate === segment || candidate.endsWith(`_${segment}`)) matches.push(candidate);
|
|
2314
|
+
}
|
|
2315
|
+
}
|
|
2316
|
+
}
|
|
2317
|
+
return matches;
|
|
2318
|
+
}
|
|
2319
|
+
|
|
2320
|
+
/**
|
|
2321
|
+
* `Tool <name> not found`, plus a suggestion when the advertised set contains a
|
|
2322
|
+
* plausible intended target. Exact wording is not a contract; the model reads it.
|
|
2323
|
+
*/
|
|
2324
|
+
function formatToolNotFoundMessage(
|
|
2325
|
+
name: string,
|
|
2326
|
+
tools: ReadonlyArray<Pick<AgentTool, "name" | "customWireName">> | undefined,
|
|
2327
|
+
): string {
|
|
2328
|
+
const suggestions = suggestToolNames(name, tools);
|
|
2329
|
+
if (suggestions.length === 0) return `Tool ${name} not found`;
|
|
2330
|
+
if (suggestions.length === 1) return `Tool ${name} not found. Did you mean ${suggestions[0]}?`;
|
|
2331
|
+
return `Tool ${name} not found. Closest available: ${suggestions.slice(0, MAX_TOOL_NAME_SUGGESTIONS).join(", ")}`;
|
|
2332
|
+
}
|
|
2333
|
+
|
|
2272
2334
|
/**
|
|
2273
2335
|
* Pre-dispatch phase for every pending tool call on `assistantMessage`, run in
|
|
2274
2336
|
* call order: intent extraction, argument validation, and the `beforeToolCall`
|
|
@@ -2312,7 +2374,7 @@ async function prepareToolCallDispatch(
|
|
|
2312
2374
|
}
|
|
2313
2375
|
const validate = (args: Record<string, unknown>): Record<string, unknown> | undefined => {
|
|
2314
2376
|
try {
|
|
2315
|
-
if (!tool) throw new Error(
|
|
2377
|
+
if (!tool) throw new Error(formatToolNotFoundMessage(toolCall.name, context.tools));
|
|
2316
2378
|
return validateToolArguments(tool, { ...toolCall, arguments: args });
|
|
2317
2379
|
} catch (validationError) {
|
|
2318
2380
|
if (tool?.lenientArgValidation) {
|
|
@@ -2637,7 +2699,7 @@ async function executeToolCalls(
|
|
|
2637
2699
|
|
|
2638
2700
|
await runInActiveSpan(toolSpan, async () => {
|
|
2639
2701
|
try {
|
|
2640
|
-
if (!tool) throw new Error(
|
|
2702
|
+
if (!tool) throw new Error(formatToolNotFoundMessage(toolCall.name, tools));
|
|
2641
2703
|
if (record.signal.aborted) {
|
|
2642
2704
|
result = createToolSignalAbortedResult(record.signal);
|
|
2643
2705
|
isError = true;
|
|
@@ -0,0 +1,310 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Anthropic server-side compaction (`compact-2026-01-12` beta).
|
|
3
|
+
*
|
|
4
|
+
* The compaction request is the live turn's own request shape — same system
|
|
5
|
+
* prompt, tools, and message history — plus the `compact_20260112` edit with
|
|
6
|
+
* `pause_after_compaction`. The API summarizes the prompt from the already
|
|
7
|
+
* cached prefix and stops; the summary arrives as a `compaction` block that
|
|
8
|
+
* the provider surfaces as an `anthropicCompaction` payload. The summary is
|
|
9
|
+
* plain text, so it doubles as the compaction entry's readable summary for
|
|
10
|
+
* every other provider, while the Anthropic provider replays it as a native
|
|
11
|
+
* block (the API drops everything that precedes it). The retained tail after
|
|
12
|
+
* the cut point is replayed from session entries exactly like a local summary.
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
import type {
|
|
16
|
+
AnthropicCompactionPayload,
|
|
17
|
+
ApiKey,
|
|
18
|
+
AssistantMessage,
|
|
19
|
+
Effort,
|
|
20
|
+
Message,
|
|
21
|
+
Model,
|
|
22
|
+
SimpleStreamOptions,
|
|
23
|
+
Tool,
|
|
24
|
+
Usage,
|
|
25
|
+
} from "@oh-my-pi/pi-ai";
|
|
26
|
+
import * as AIError from "@oh-my-pi/pi-ai/error";
|
|
27
|
+
import { supportsAnthropicCompaction } from "@oh-my-pi/pi-ai/providers/anthropic";
|
|
28
|
+
import { isRecord, prompt } from "@oh-my-pi/pi-utils";
|
|
29
|
+
import { type InstrumentedChatSpanOptions, instrumentedCompleteSimple } from "../telemetry";
|
|
30
|
+
import anthropicCompactionInstructionsPrompt from "./prompts/anthropic-compaction-instructions.md" with { type: "text" };
|
|
31
|
+
|
|
32
|
+
export const ANTHROPIC_COMPACTION_PRESERVE_KEY = "anthropicCompaction";
|
|
33
|
+
|
|
34
|
+
/** The API rejects a `compact_20260112` trigger below this many input tokens. */
|
|
35
|
+
export const ANTHROPIC_COMPACTION_MIN_TRIGGER_TOKENS = 50_000;
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* Smallest context the native lane accepts. The trigger sits at the API
|
|
39
|
+
* floor, so a prompt that lands below it is answered instead of compacted;
|
|
40
|
+
* the margin over the floor absorbs the difference between the last reported
|
|
41
|
+
* context size and the compaction request's own input.
|
|
42
|
+
*/
|
|
43
|
+
export const ANTHROPIC_COMPACTION_MIN_CONTEXT_TOKENS = 55_000;
|
|
44
|
+
|
|
45
|
+
/** Summary persisted under {@link ANTHROPIC_COMPACTION_PRESERVE_KEY}. */
|
|
46
|
+
export interface AnthropicCompactionPreserveData {
|
|
47
|
+
provider: string;
|
|
48
|
+
content: string;
|
|
49
|
+
/** Opaque provider state the API attached to the block; replayed verbatim. */
|
|
50
|
+
encryptedContent?: string;
|
|
51
|
+
/** Harness file metadata (`<files>` section) replayed after the native block. */
|
|
52
|
+
filesText?: string;
|
|
53
|
+
/** Model that wrote the summary. */
|
|
54
|
+
model?: string;
|
|
55
|
+
/** Prompt tokens the compaction request processed, for display. */
|
|
56
|
+
usedTokens?: number;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
function isAnthropicMessagesModel(model: Model): model is Model<"anthropic-messages"> {
|
|
60
|
+
return model.api === "anthropic-messages";
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Whether a model compacts through the Anthropic compaction beta. Model
|
|
65
|
+
* eligibility is catalog policy (`compat.supportsServerCompaction`, the
|
|
66
|
+
* lineage the beta documents); endpoint eligibility is resolved the way the
|
|
67
|
+
* provider routes requests, so a Foundry or `ANTHROPIC_BASE_URL` reroute of a
|
|
68
|
+
* first-party model is excluded unless the route opted in with
|
|
69
|
+
* `remoteCompaction.enabled`.
|
|
70
|
+
*/
|
|
71
|
+
export function shouldUseAnthropicNativeCompaction(model: Model): model is Model<"anthropic-messages"> {
|
|
72
|
+
return isAnthropicMessagesModel(model) && supportsAnthropicCompaction(model);
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
export function getPreservedAnthropicCompactionData(
|
|
76
|
+
preserveData: Record<string, unknown> | undefined,
|
|
77
|
+
): AnthropicCompactionPreserveData | undefined {
|
|
78
|
+
const candidate = preserveData?.[ANTHROPIC_COMPACTION_PRESERVE_KEY];
|
|
79
|
+
if (!isRecord(candidate)) return undefined;
|
|
80
|
+
if (typeof candidate.provider !== "string" || candidate.provider.length === 0) return undefined;
|
|
81
|
+
if (typeof candidate.content !== "string" || candidate.content.length === 0) return undefined;
|
|
82
|
+
return {
|
|
83
|
+
provider: candidate.provider,
|
|
84
|
+
content: candidate.content,
|
|
85
|
+
...(typeof candidate.encryptedContent === "string" && candidate.encryptedContent.length > 0
|
|
86
|
+
? { encryptedContent: candidate.encryptedContent }
|
|
87
|
+
: {}),
|
|
88
|
+
...(typeof candidate.filesText === "string" && candidate.filesText.length > 0
|
|
89
|
+
? { filesText: candidate.filesText }
|
|
90
|
+
: {}),
|
|
91
|
+
...(typeof candidate.model === "string" ? { model: candidate.model } : {}),
|
|
92
|
+
...(typeof candidate.usedTokens === "number" ? { usedTokens: candidate.usedTokens } : {}),
|
|
93
|
+
};
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
/** Set or strip the Anthropic compaction slot; a new compaction never inherits a stale summary. */
|
|
97
|
+
export function withAnthropicCompactionPreserveData(
|
|
98
|
+
preserveData: Record<string, unknown> | undefined,
|
|
99
|
+
compaction: AnthropicCompactionPreserveData | undefined,
|
|
100
|
+
): Record<string, unknown> | undefined {
|
|
101
|
+
if (compaction) {
|
|
102
|
+
return { ...preserveData, [ANTHROPIC_COMPACTION_PRESERVE_KEY]: compaction };
|
|
103
|
+
}
|
|
104
|
+
if (!preserveData || !(ANTHROPIC_COMPACTION_PRESERVE_KEY in preserveData)) {
|
|
105
|
+
return preserveData;
|
|
106
|
+
}
|
|
107
|
+
const { [ANTHROPIC_COMPACTION_PRESERVE_KEY]: _removed, ...rest } = preserveData;
|
|
108
|
+
return Object.keys(rest).length > 0 ? rest : undefined;
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
/** Replay payload for a compaction summary the active model produced natively. */
|
|
112
|
+
export function getAnthropicCompactionPayload(
|
|
113
|
+
preserveData: Record<string, unknown> | undefined,
|
|
114
|
+
): AnthropicCompactionPayload | undefined {
|
|
115
|
+
const preserved = getPreservedAnthropicCompactionData(preserveData);
|
|
116
|
+
if (!preserved) return undefined;
|
|
117
|
+
return {
|
|
118
|
+
type: "anthropicCompaction",
|
|
119
|
+
provider: preserved.provider,
|
|
120
|
+
content: preserved.content,
|
|
121
|
+
...(preserved.encryptedContent ? { encryptedContent: preserved.encryptedContent } : {}),
|
|
122
|
+
...(preserved.filesText ? { filesText: preserved.filesText } : {}),
|
|
123
|
+
};
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/**
|
|
127
|
+
* The retained tail as the model will see it, for the summarization
|
|
128
|
+
* instructions: how many of the conversation's final wire messages stay in
|
|
129
|
+
* context verbatim, and the role of the first. The compaction request carries
|
|
130
|
+
* the whole conversation so the prompt cache the live turn wrote is read, but
|
|
131
|
+
* the summary must cover only the history before that tail — the local
|
|
132
|
+
* summarizer never sees the tail, and the rebuilt context replays it after the
|
|
133
|
+
* summary. Counting mirrors the provider's message conversion (consecutive
|
|
134
|
+
* tool results collapse into one user message; developer messages are user
|
|
135
|
+
* messages). Structured for the prompt template, which renders the
|
|
136
|
+
* singular/plural wording; the description quotes no content: quoting the
|
|
137
|
+
* tail would hand the summarizer the very facts it must leave to the tail.
|
|
138
|
+
*/
|
|
139
|
+
export interface RetainedTailScope {
|
|
140
|
+
count: number;
|
|
141
|
+
role: "assistant" | "user";
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
export function describeRetainedTail(messages: readonly Message[]): RetainedTailScope | undefined {
|
|
145
|
+
const first = messages[0];
|
|
146
|
+
if (!first) return undefined;
|
|
147
|
+
let count = 0;
|
|
148
|
+
let previousWasToolResult = false;
|
|
149
|
+
for (const message of messages) {
|
|
150
|
+
const isToolResult = message.role === "toolResult";
|
|
151
|
+
if (!(isToolResult && previousWasToolResult)) count += 1;
|
|
152
|
+
previousWasToolResult = isToolResult;
|
|
153
|
+
}
|
|
154
|
+
// Mirror the provider's trailing-assistant prefill: a tail ending in a
|
|
155
|
+
// live assistant turn gains a synthetic trailing user message on the
|
|
156
|
+
// wire, which stays verbatim too. Only blocks the converter emits count —
|
|
157
|
+
// blank text never serializes, and images, redacted thinking, and fallback
|
|
158
|
+
// markers need target context this scope lacks, so a turn of only those
|
|
159
|
+
// emits nothing and draws no pad. Without the pad in
|
|
160
|
+
// the scope, the summary could duplicate the tail head.
|
|
161
|
+
const last = messages[messages.length - 1];
|
|
162
|
+
if (last?.role === "assistant" && last.content.some(emitsWireBlock)) {
|
|
163
|
+
count += 1;
|
|
164
|
+
}
|
|
165
|
+
return { count, role: first.role === "assistant" ? "assistant" : "user" };
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/**
|
|
169
|
+
* Whether an assistant content block reaches the wire. Server-tool blocks
|
|
170
|
+
* serialize unconditionally; blank text never does. Redacted thinking and
|
|
171
|
+
* fallback markers replay only for specific deployments, which the scope
|
|
172
|
+
* cannot see, so a turn of only those conservatively draws no pad.
|
|
173
|
+
*/
|
|
174
|
+
function emitsWireBlock(block: AssistantMessage["content"][number]): boolean {
|
|
175
|
+
switch (block.type) {
|
|
176
|
+
case "text":
|
|
177
|
+
return block.text.trim().length > 0;
|
|
178
|
+
case "toolCall":
|
|
179
|
+
case "anthropicServerTool":
|
|
180
|
+
return true;
|
|
181
|
+
case "thinking":
|
|
182
|
+
return block.thinking.trim().length > 0 || (block.thinkingSignature ?? "").trim().length > 0;
|
|
183
|
+
default:
|
|
184
|
+
return false;
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* Summarization prompt sent as the edit's `instructions`, which replace the
|
|
190
|
+
* API default entirely. The template lays out the retained-tail boundary
|
|
191
|
+
* first, so the summary covers only the history the rebuilt context drops,
|
|
192
|
+
* then the caller's extra context, the same structure prompt as the local
|
|
193
|
+
* summarizer, the caller's focus, and the tool-abstention clause the API
|
|
194
|
+
* recommends when tools are defined (a summarization pass that calls a tool
|
|
195
|
+
* yields no summary).
|
|
196
|
+
*/
|
|
197
|
+
export function buildAnthropicCompactionInstructions(
|
|
198
|
+
basePrompt: string,
|
|
199
|
+
customInstructions: string | undefined,
|
|
200
|
+
extraContext: string | undefined,
|
|
201
|
+
retainedTail: RetainedTailScope | undefined,
|
|
202
|
+
): string {
|
|
203
|
+
return prompt.render(anthropicCompactionInstructionsPrompt, {
|
|
204
|
+
basePrompt,
|
|
205
|
+
customInstructions,
|
|
206
|
+
extraContext,
|
|
207
|
+
retainedTail,
|
|
208
|
+
});
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
export interface AnthropicNativeCompactionRequest {
|
|
212
|
+
systemPrompt: string[];
|
|
213
|
+
messages: Message[];
|
|
214
|
+
tools?: Tool[];
|
|
215
|
+
instructions: string;
|
|
216
|
+
maxTokens: number;
|
|
217
|
+
reasoning?: Effort;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
export interface AnthropicNativeCompactionResponse {
|
|
221
|
+
content: string;
|
|
222
|
+
encryptedContent?: string;
|
|
223
|
+
usage: Usage;
|
|
224
|
+
model: string;
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
export interface AnthropicNativeCompactionOptions
|
|
228
|
+
extends
|
|
229
|
+
Pick<
|
|
230
|
+
SimpleStreamOptions,
|
|
231
|
+
| "initiatorOverride"
|
|
232
|
+
| "metadata"
|
|
233
|
+
| "fetch"
|
|
234
|
+
| "sessionId"
|
|
235
|
+
| "promptCacheKey"
|
|
236
|
+
| "providerSessionState"
|
|
237
|
+
| "maxInFlightRequests"
|
|
238
|
+
>,
|
|
239
|
+
Pick<InstrumentedChatSpanOptions, "completeImpl" | "telemetry" | "retry"> {}
|
|
240
|
+
|
|
241
|
+
/**
|
|
242
|
+
* Run one compaction request and return the summary the API wrote, with the
|
|
243
|
+
* opaque `encrypted_content` the API attached for the replay. `completeSimple`
|
|
244
|
+
* resolves terminal failures as messages, so their classification is restored
|
|
245
|
+
* here: an aborted response is an `AbortError` (a cancellation, never a native
|
|
246
|
+
* failure) and an error response keeps its HTTP status, so auth and timeout
|
|
247
|
+
* handling downstream classify it the same way as the OpenAI lanes. A response
|
|
248
|
+
* without a summary is a native failure — the API answers the prompt instead
|
|
249
|
+
* when its input never reached the trigger, and returns an empty block when
|
|
250
|
+
* the model called a tool during summarization.
|
|
251
|
+
*/
|
|
252
|
+
export async function requestAnthropicNativeCompaction(
|
|
253
|
+
model: Model<"anthropic-messages">,
|
|
254
|
+
apiKey: ApiKey,
|
|
255
|
+
request: AnthropicNativeCompactionRequest,
|
|
256
|
+
signal: AbortSignal | undefined,
|
|
257
|
+
options: AnthropicNativeCompactionOptions,
|
|
258
|
+
): Promise<AnthropicNativeCompactionResponse> {
|
|
259
|
+
const response = await instrumentedCompleteSimple(
|
|
260
|
+
model,
|
|
261
|
+
{ systemPrompt: request.systemPrompt, messages: request.messages, tools: request.tools },
|
|
262
|
+
{
|
|
263
|
+
apiKey,
|
|
264
|
+
signal,
|
|
265
|
+
maxTokens: request.maxTokens,
|
|
266
|
+
reasoning: request.reasoning,
|
|
267
|
+
initiatorOverride: options.initiatorOverride,
|
|
268
|
+
metadata: options.metadata,
|
|
269
|
+
fetch: options.fetch,
|
|
270
|
+
sessionId: options.sessionId,
|
|
271
|
+
promptCacheKey: options.promptCacheKey,
|
|
272
|
+
providerSessionState: options.providerSessionState,
|
|
273
|
+
maxInFlightRequests: options.maxInFlightRequests,
|
|
274
|
+
anthropicCompaction: {
|
|
275
|
+
triggerInputTokens: ANTHROPIC_COMPACTION_MIN_TRIGGER_TOKENS,
|
|
276
|
+
pauseAfterCompaction: true,
|
|
277
|
+
instructions: request.instructions,
|
|
278
|
+
},
|
|
279
|
+
},
|
|
280
|
+
{
|
|
281
|
+
telemetry: options.telemetry,
|
|
282
|
+
oneshotKind: "compaction_native",
|
|
283
|
+
completeImpl: options.completeImpl,
|
|
284
|
+
retry: options.retry,
|
|
285
|
+
},
|
|
286
|
+
);
|
|
287
|
+
if (response.stopReason === "aborted") {
|
|
288
|
+
throw new AIError.AbortError("Anthropic compaction aborted", { cause: signal?.reason });
|
|
289
|
+
}
|
|
290
|
+
if (response.stopReason === "error") {
|
|
291
|
+
const message = `Anthropic compaction failed: ${response.errorMessage ?? "unknown error"}`;
|
|
292
|
+
throw response.errorStatus === undefined
|
|
293
|
+
? new Error(message)
|
|
294
|
+
: new AIError.ProviderHttpError(message, response.errorStatus);
|
|
295
|
+
}
|
|
296
|
+
const payload = response.providerPayload;
|
|
297
|
+
if (payload?.type !== "anthropicCompaction" || payload.content.length === 0) {
|
|
298
|
+
throw new Error(
|
|
299
|
+
response.stopDetails?.type === "compaction"
|
|
300
|
+
? "Anthropic compaction returned no summary"
|
|
301
|
+
: "Anthropic compaction response carried no compaction block",
|
|
302
|
+
);
|
|
303
|
+
}
|
|
304
|
+
return {
|
|
305
|
+
content: payload.content,
|
|
306
|
+
encryptedContent: payload.encryptedContent,
|
|
307
|
+
usage: response.usage,
|
|
308
|
+
model: response.model,
|
|
309
|
+
};
|
|
310
|
+
}
|
|
@@ -44,8 +44,8 @@ export const V2_RETAINED_MESSAGE_TOKEN_BUDGET = 64_000;
|
|
|
44
44
|
/** Max retries for V2 streaming compaction on transient stream errors. */
|
|
45
45
|
export const V2_COMPACTION_MAX_RETRIES = 2;
|
|
46
46
|
|
|
47
|
-
/** Timeout for V2 streaming compaction (
|
|
48
|
-
export const V2_COMPACTION_TIMEOUT_MS =
|
|
47
|
+
/** Timeout for V2 streaming compaction (5 minutes, same as V1). */
|
|
48
|
+
export const V2_COMPACTION_TIMEOUT_MS = 300_000;
|
|
49
49
|
|
|
50
50
|
const DEFAULT_AZURE_API_VERSION = "v1";
|
|
51
51
|
const OPENAI_REMOTE_COMPACTION_PRESERVE_KEY = "openaiRemoteCompaction";
|
|
@@ -42,6 +42,15 @@ import { type AgentTelemetry, instrumentedCompleteSimple } from "../telemetry";
|
|
|
42
42
|
import { ThinkingLevel } from "../thinking";
|
|
43
43
|
import { Tokenizer } from "../tokenizer";
|
|
44
44
|
import type { AgentMessage } from "../types";
|
|
45
|
+
import {
|
|
46
|
+
ANTHROPIC_COMPACTION_MIN_CONTEXT_TOKENS,
|
|
47
|
+
buildAnthropicCompactionInstructions,
|
|
48
|
+
describeRetainedTail,
|
|
49
|
+
getPreservedAnthropicCompactionData,
|
|
50
|
+
requestAnthropicNativeCompaction,
|
|
51
|
+
shouldUseAnthropicNativeCompaction,
|
|
52
|
+
withAnthropicCompactionPreserveData,
|
|
53
|
+
} from "./anthropic";
|
|
45
54
|
import {
|
|
46
55
|
buildCompactionV2Request,
|
|
47
56
|
buildCompactionV2RequestFromBody,
|
|
@@ -53,7 +62,13 @@ import {
|
|
|
53
62
|
} from "./compaction-v2-streaming";
|
|
54
63
|
import type { CompactionEntry, SessionEntry } from "./entries";
|
|
55
64
|
import { NativeCompactionError } from "./errors";
|
|
56
|
-
import {
|
|
65
|
+
import {
|
|
66
|
+
type ConvertToLlm,
|
|
67
|
+
createBranchSummaryMessage,
|
|
68
|
+
createCompactionSummaryMessage,
|
|
69
|
+
createCustomMessage,
|
|
70
|
+
defaultConvertToLlm,
|
|
71
|
+
} from "./messages";
|
|
57
72
|
import {
|
|
58
73
|
buildOpenAiNativeHistory,
|
|
59
74
|
getPreservedOpenAiRemoteCompactionData,
|
|
@@ -224,7 +239,11 @@ export const DEFAULT_COMPACTION_SETTINGS: CompactionSettings = {
|
|
|
224
239
|
v2RetainedMessageBudget: V2_RETAINED_MESSAGE_TOKEN_BUDGET,
|
|
225
240
|
};
|
|
226
241
|
|
|
227
|
-
/**
|
|
242
|
+
/**
|
|
243
|
+
* Whether a compaction candidate preserves provider-native transport under the
|
|
244
|
+
* effective settings: an OpenAI Responses compact route (V1 or streamed V2) or
|
|
245
|
+
* the Anthropic compaction beta.
|
|
246
|
+
*/
|
|
228
247
|
export function shouldUseProviderNativeCompaction(
|
|
229
248
|
model: Model,
|
|
230
249
|
settings: Pick<CompactionSettings, "remoteEnabled" | "remoteStreamingV2Enabled">,
|
|
@@ -232,7 +251,8 @@ export function shouldUseProviderNativeCompaction(
|
|
|
232
251
|
if (settings.remoteEnabled === false) return false;
|
|
233
252
|
return (
|
|
234
253
|
shouldUseOpenAiRemoteCompaction(model) ||
|
|
235
|
-
(settings.remoteStreamingV2Enabled !== false && shouldUseCompactionV2Streaming(model))
|
|
254
|
+
(settings.remoteStreamingV2Enabled !== false && shouldUseCompactionV2Streaming(model)) ||
|
|
255
|
+
shouldUseAnthropicNativeCompaction(model)
|
|
236
256
|
);
|
|
237
257
|
}
|
|
238
258
|
|
|
@@ -395,22 +415,6 @@ export function resolveThresholdTokens(contextWindow: number, settings: Compacti
|
|
|
395
415
|
// Cut point detection
|
|
396
416
|
// ============================================================================
|
|
397
417
|
|
|
398
|
-
function estimateEntriesTokens(
|
|
399
|
-
entries: SessionEntry[],
|
|
400
|
-
tokenizer: Tokenizer,
|
|
401
|
-
startIndex: number,
|
|
402
|
-
endIndex: number,
|
|
403
|
-
): number {
|
|
404
|
-
let total = 0;
|
|
405
|
-
for (let i = startIndex; i < endIndex; i++) {
|
|
406
|
-
const msg = getMessageFromEntry(entries[i]);
|
|
407
|
-
if (msg) {
|
|
408
|
-
total += tokenizer.countMessage(msg);
|
|
409
|
-
}
|
|
410
|
-
}
|
|
411
|
-
return total;
|
|
412
|
-
}
|
|
413
|
-
|
|
414
418
|
/**
|
|
415
419
|
* Find valid cut points: indices of user, assistant, custom, or bashExecution messages.
|
|
416
420
|
* Never cut at tool results (they must follow their tool call).
|
|
@@ -1252,6 +1256,8 @@ export interface CompactionPreparation {
|
|
|
1252
1256
|
tokensBefore: number;
|
|
1253
1257
|
/** Summary from previous compaction, for iterative update */
|
|
1254
1258
|
previousSummary?: string;
|
|
1259
|
+
/** ISO timestamp of the previous compaction entry, for iterative update */
|
|
1260
|
+
previousSummaryTimestamp?: string;
|
|
1255
1261
|
/** Preserved opaque compaction payload from the previous compaction, if any. */
|
|
1256
1262
|
previousPreserveData?: Record<string, unknown>;
|
|
1257
1263
|
/** File operations extracted from messagesToSummarize */
|
|
@@ -1331,16 +1337,10 @@ export function prepareCompaction(
|
|
|
1331
1337
|
|
|
1332
1338
|
let prevCompactionIndex = findReadableCompactionIndex(pathEntries, settings, activeModel);
|
|
1333
1339
|
|
|
1334
|
-
//
|
|
1335
|
-
//
|
|
1336
|
-
// must not resurrect the dropped pre-clear turns into its summary — matching
|
|
1337
|
-
// how buildSessionContext starts the model-context rebuild after the boundary.
|
|
1338
|
-
// A boundary after the last reusable compaction supersedes it: the pre-reset
|
|
1339
|
-
// summary was cleared too, so drop the previous-compaction reuse and start
|
|
1340
|
-
// fresh after the boundary. A boundary at or before that compaction is already
|
|
1341
|
-
// superseded by it, so only scan newer entries.
|
|
1340
|
+
// A reset after the reusable compaction clears its summary too. An older
|
|
1341
|
+
// reset still bounds how far we may recover that compaction's kept messages.
|
|
1342
1342
|
let resetBoundaryIndex = -1;
|
|
1343
|
-
for (let i = pathEntries.length - 1; i
|
|
1343
|
+
for (let i = pathEntries.length - 1; i >= 0; i--) {
|
|
1344
1344
|
if (pathEntries[i].type === "reset_boundary") {
|
|
1345
1345
|
resetBoundaryIndex = i;
|
|
1346
1346
|
break;
|
|
@@ -1349,14 +1349,43 @@ export function prepareCompaction(
|
|
|
1349
1349
|
if (resetBoundaryIndex > prevCompactionIndex) {
|
|
1350
1350
|
prevCompactionIndex = -1;
|
|
1351
1351
|
}
|
|
1352
|
-
const
|
|
1353
|
-
|
|
1352
|
+
const previousCompaction =
|
|
1353
|
+
prevCompactionIndex >= 0 ? (pathEntries[prevCompactionIndex] as CompactionEntry) : undefined;
|
|
1354
|
+
let boundaryStart = Math.max(prevCompactionIndex, resetBoundaryIndex) + 1;
|
|
1355
|
+
if (
|
|
1356
|
+
previousCompaction &&
|
|
1357
|
+
!getCompactionV2PreserveData(previousCompaction.preserveData) &&
|
|
1358
|
+
!getPreservedOpenAiRemoteCompactionData(previousCompaction.preserveData)
|
|
1359
|
+
) {
|
|
1360
|
+
// Local summaries exclude the retained tail, whose original entries precede
|
|
1361
|
+
// the compaction record. Native replay already carries that tail. Only look
|
|
1362
|
+
// backwards: advisor snapshots put all retained messages after the summary
|
|
1363
|
+
// and may carry a keep ID from their previous, differently indexed snapshot.
|
|
1364
|
+
for (let i = resetBoundaryIndex + 1; i < prevCompactionIndex; i++) {
|
|
1365
|
+
if (pathEntries[i].id === previousCompaction.firstKeptEntryId) {
|
|
1366
|
+
boundaryStart = i;
|
|
1367
|
+
break;
|
|
1368
|
+
}
|
|
1369
|
+
}
|
|
1370
|
+
}
|
|
1371
|
+
|
|
1372
|
+
// Keep original IDs beside the converted messages so estimation, cutting,
|
|
1373
|
+
// and all three output regions share one sequence without journal metadata.
|
|
1374
|
+
const compactionEntries: SessionEntry[] = [];
|
|
1375
|
+
const compactionMessages: AgentMessage[] = [];
|
|
1376
|
+
for (let i = boundaryStart; i < pathEntries.length; i++) {
|
|
1377
|
+
const entry = pathEntries[i];
|
|
1378
|
+
const message = getMessageFromEntry(entry);
|
|
1379
|
+
if (!message) continue;
|
|
1380
|
+
compactionEntries.push(entry);
|
|
1381
|
+
compactionMessages.push(message);
|
|
1382
|
+
}
|
|
1354
1383
|
|
|
1355
1384
|
const lastUsage = getLastAssistantUsage(pathEntries);
|
|
1356
1385
|
const tokensBefore = lastUsage ? calculateContextTokens(lastUsage) : 0;
|
|
1357
1386
|
let keepRecentTokens = settings.keepRecentTokens;
|
|
1358
1387
|
if (lastUsage) {
|
|
1359
|
-
const estimatedTokens =
|
|
1388
|
+
const estimatedTokens = tokenizer.countMessages(compactionMessages);
|
|
1360
1389
|
const promptTokens = calculatePromptTokens(lastUsage);
|
|
1361
1390
|
const ratio = estimatedTokens > 0 ? promptTokens / estimatedTokens : 0;
|
|
1362
1391
|
if (Number.isFinite(ratio) && ratio > 1) {
|
|
@@ -1364,10 +1393,10 @@ export function prepareCompaction(
|
|
|
1364
1393
|
}
|
|
1365
1394
|
}
|
|
1366
1395
|
|
|
1367
|
-
const cutPoint = findCutPoint(
|
|
1396
|
+
const cutPoint = findCutPoint(compactionEntries, tokenizer, 0, compactionEntries.length, keepRecentTokens);
|
|
1368
1397
|
|
|
1369
1398
|
// Get ID of first kept entry
|
|
1370
|
-
const firstKeptEntry =
|
|
1399
|
+
const firstKeptEntry = compactionEntries[cutPoint.firstKeptEntryIndex];
|
|
1371
1400
|
if (!firstKeptEntry?.id) {
|
|
1372
1401
|
return undefined; // Session needs migration
|
|
1373
1402
|
}
|
|
@@ -1375,42 +1404,16 @@ export function prepareCompaction(
|
|
|
1375
1404
|
|
|
1376
1405
|
const historyEnd = cutPoint.isSplitTurn ? cutPoint.turnStartIndex : cutPoint.firstKeptEntryIndex;
|
|
1377
1406
|
|
|
1378
|
-
|
|
1379
|
-
const
|
|
1380
|
-
|
|
1381
|
-
|
|
1382
|
-
|
|
1383
|
-
}
|
|
1384
|
-
|
|
1385
|
-
// Messages for turn prefix summary (if splitting a turn)
|
|
1386
|
-
const turnPrefixMessages: AgentMessage[] = [];
|
|
1387
|
-
if (cutPoint.isSplitTurn) {
|
|
1388
|
-
for (let i = cutPoint.turnStartIndex; i < cutPoint.firstKeptEntryIndex; i++) {
|
|
1389
|
-
const msg = getMessageFromEntry(pathEntries[i]);
|
|
1390
|
-
if (msg) turnPrefixMessages.push(msg);
|
|
1391
|
-
}
|
|
1392
|
-
}
|
|
1393
|
-
|
|
1394
|
-
// Messages kept after compaction (recent history)
|
|
1395
|
-
const recentMessages: AgentMessage[] = [];
|
|
1396
|
-
for (let i = cutPoint.firstKeptEntryIndex; i < boundaryEnd; i++) {
|
|
1397
|
-
const msg = getMessageFromEntry(pathEntries[i]);
|
|
1398
|
-
if (msg) recentMessages.push(msg);
|
|
1399
|
-
}
|
|
1407
|
+
const messagesToSummarize = compactionMessages.slice(0, historyEnd);
|
|
1408
|
+
const turnPrefixMessages = cutPoint.isSplitTurn
|
|
1409
|
+
? compactionMessages.slice(cutPoint.turnStartIndex, cutPoint.firstKeptEntryIndex)
|
|
1410
|
+
: [];
|
|
1411
|
+
const recentMessages = compactionMessages.slice(cutPoint.firstKeptEntryIndex);
|
|
1400
1412
|
// Nothing to summarize means compaction would be a no-op.
|
|
1401
1413
|
if (messagesToSummarize.length === 0 && turnPrefixMessages.length === 0) {
|
|
1402
1414
|
return undefined;
|
|
1403
1415
|
}
|
|
1404
1416
|
|
|
1405
|
-
// Get previous summary and preserved data for iterative updates
|
|
1406
|
-
let previousSummary: string | undefined;
|
|
1407
|
-
let previousPreserveData: Record<string, unknown> | undefined;
|
|
1408
|
-
if (prevCompactionIndex >= 0) {
|
|
1409
|
-
const prevCompaction = pathEntries[prevCompactionIndex] as CompactionEntry;
|
|
1410
|
-
previousSummary = prevCompaction.summary;
|
|
1411
|
-
previousPreserveData = prevCompaction.preserveData;
|
|
1412
|
-
}
|
|
1413
|
-
|
|
1414
1417
|
// Extract file operations from messages and previous compaction
|
|
1415
1418
|
const fileOps = extractFileOperations(messagesToSummarize, pathEntries, prevCompactionIndex);
|
|
1416
1419
|
|
|
@@ -1428,8 +1431,9 @@ export function prepareCompaction(
|
|
|
1428
1431
|
recentMessages,
|
|
1429
1432
|
isSplitTurn: cutPoint.isSplitTurn,
|
|
1430
1433
|
tokensBefore,
|
|
1431
|
-
previousSummary,
|
|
1432
|
-
|
|
1434
|
+
previousSummary: previousCompaction?.summary,
|
|
1435
|
+
previousSummaryTimestamp: previousCompaction?.timestamp,
|
|
1436
|
+
previousPreserveData: previousCompaction?.preserveData,
|
|
1433
1437
|
fileOps,
|
|
1434
1438
|
settings,
|
|
1435
1439
|
};
|
|
@@ -1585,6 +1589,9 @@ export async function compact(
|
|
|
1585
1589
|
tools: options?.tools,
|
|
1586
1590
|
fetch: options?.fetch,
|
|
1587
1591
|
completeImpl: options?.completeImpl,
|
|
1592
|
+
// The caller's opt-out must reach every summarization oneshot, otherwise
|
|
1593
|
+
// an outer retry loop multiplies with the inner one (see SummaryOptions).
|
|
1594
|
+
oneshotRetry: options?.oneshotRetry,
|
|
1588
1595
|
};
|
|
1589
1596
|
|
|
1590
1597
|
const previousSnapcompactArchive = snapcompact.getPreservedArchive(previousPreserveData);
|
|
@@ -1599,7 +1606,10 @@ export async function compact(
|
|
|
1599
1606
|
? createSnapcompactArchiveMigrationMessage(previousSnapcompactArchiveText)
|
|
1600
1607
|
: undefined;
|
|
1601
1608
|
|
|
1602
|
-
let preserveData =
|
|
1609
|
+
let preserveData = withAnthropicCompactionPreserveData(
|
|
1610
|
+
withOpenAiRemoteCompactionPreserveData(previousPreserveData, undefined),
|
|
1611
|
+
undefined,
|
|
1612
|
+
);
|
|
1603
1613
|
const remoteMessages: AgentMessage[] = [
|
|
1604
1614
|
...(snapcompactArchiveMigrationMessage ? [snapcompactArchiveMigrationMessage] : []),
|
|
1605
1615
|
...messagesToSummarize,
|
|
@@ -1784,6 +1794,112 @@ export async function compact(
|
|
|
1784
1794
|
}
|
|
1785
1795
|
}
|
|
1786
1796
|
|
|
1797
|
+
// Anthropic server-side compaction: the live turn's request shape plus the
|
|
1798
|
+
// compact edit, so the API summarizes from its cached prefix. The summary is
|
|
1799
|
+
// real text, persisted both as the entry summary and as the native replay
|
|
1800
|
+
// payload. A context below the API's trigger floor cannot compact remotely
|
|
1801
|
+
// and takes the local summarizer instead — an eligibility boundary, not a
|
|
1802
|
+
// failure.
|
|
1803
|
+
let nativeSummary: string | undefined;
|
|
1804
|
+
let nativeEncryptedContent: string | undefined;
|
|
1805
|
+
let nativeUsedTokens: number | undefined;
|
|
1806
|
+
if (
|
|
1807
|
+
!usedRemoteCompaction &&
|
|
1808
|
+
settings.remoteEnabled !== false &&
|
|
1809
|
+
shouldUseAnthropicNativeCompaction(model) &&
|
|
1810
|
+
tokensBefore >= ANTHROPIC_COMPACTION_MIN_CONTEXT_TOKENS
|
|
1811
|
+
) {
|
|
1812
|
+
const previousNative = getPreservedAnthropicCompactionData(previousPreserveData);
|
|
1813
|
+
// Lead with the previous summary exactly as the live context renders it:
|
|
1814
|
+
// natively when this provider wrote it, as text otherwise. The request
|
|
1815
|
+
// then shares the live turn's prefix byte-for-byte. A prior snapcompact
|
|
1816
|
+
// archive is already merged into that summary text, so the archive
|
|
1817
|
+
// migration message the OpenAI lanes carry is omitted here.
|
|
1818
|
+
// The rewrite marker must precede every message this request replays —
|
|
1819
|
+
// summarized history and retained tail alike — exactly like the live
|
|
1820
|
+
// context rebuild predates its tail. A previous compaction's commit
|
|
1821
|
+
// timestamp is newer than re-retained or re-summarized turns, so
|
|
1822
|
+
// reusing it would strip their bound thinking only in this request,
|
|
1823
|
+
// diverging from the cached live prefix (and possibly dropping below
|
|
1824
|
+
// the trigger). Manually built preparations with no input at all fall
|
|
1825
|
+
// back to the current time.
|
|
1826
|
+
const firstReplayed = messagesToSummarize[0] ?? turnPrefixMessages[0] ?? recentMessages[0];
|
|
1827
|
+
const previousSummaryAt =
|
|
1828
|
+
firstReplayed !== undefined ? new Date(firstReplayed.timestamp - 1).toISOString() : new Date().toISOString();
|
|
1829
|
+
const previousSummaryMessage = previousSummaryForCompaction
|
|
1830
|
+
? createCompactionSummaryMessage(previousSummaryForCompaction, tokensBefore, previousSummaryAt, {
|
|
1831
|
+
providerPayload:
|
|
1832
|
+
previousNative?.provider === model.provider
|
|
1833
|
+
? {
|
|
1834
|
+
type: "anthropicCompaction",
|
|
1835
|
+
provider: previousNative.provider,
|
|
1836
|
+
content: previousNative.content,
|
|
1837
|
+
...(previousNative.encryptedContent
|
|
1838
|
+
? { encryptedContent: previousNative.encryptedContent }
|
|
1839
|
+
: {}),
|
|
1840
|
+
...(previousNative.filesText ? { filesText: previousNative.filesText } : {}),
|
|
1841
|
+
}
|
|
1842
|
+
: undefined,
|
|
1843
|
+
})
|
|
1844
|
+
: undefined;
|
|
1845
|
+
const convertToLlm = summaryOptions.convertToLlm ?? defaultConvertToLlm;
|
|
1846
|
+
const retainedTail = convertToLlm(recentMessages);
|
|
1847
|
+
const messages = [
|
|
1848
|
+
...convertToLlm([
|
|
1849
|
+
...(previousSummaryMessage ? [previousSummaryMessage] : []),
|
|
1850
|
+
...messagesToSummarize,
|
|
1851
|
+
...turnPrefixMessages,
|
|
1852
|
+
]),
|
|
1853
|
+
...retainedTail,
|
|
1854
|
+
];
|
|
1855
|
+
try {
|
|
1856
|
+
const remote = await requestAnthropicNativeCompaction(
|
|
1857
|
+
model,
|
|
1858
|
+
apiKey,
|
|
1859
|
+
{
|
|
1860
|
+
systemPrompt: summaryOptions.remoteSystemPrompt ?? [SUMMARIZATION_SYSTEM_PROMPT],
|
|
1861
|
+
messages,
|
|
1862
|
+
tools: summaryOptions.tools,
|
|
1863
|
+
instructions: buildAnthropicCompactionInstructions(
|
|
1864
|
+
summaryOptions.promptOverride ?? SUMMARIZATION_PROMPT,
|
|
1865
|
+
customInstructions,
|
|
1866
|
+
formatAdditionalContext(summaryOptions.extraContext).trim() || undefined,
|
|
1867
|
+
describeRetainedTail(retainedTail),
|
|
1868
|
+
),
|
|
1869
|
+
maxTokens: Math.min(Math.floor(0.8 * reserveTokens), MAX_SUMMARY_TOKENS),
|
|
1870
|
+
reasoning: resolveCompactionEffort(model, summaryOptions.thinkingLevel),
|
|
1871
|
+
},
|
|
1872
|
+
signal,
|
|
1873
|
+
{
|
|
1874
|
+
initiatorOverride: summaryOptions.initiatorOverride,
|
|
1875
|
+
metadata: summaryOptions.metadata,
|
|
1876
|
+
fetch: summaryOptions.fetch,
|
|
1877
|
+
sessionId: summaryOptions.sessionId,
|
|
1878
|
+
promptCacheKey: summaryOptions.promptCacheKey,
|
|
1879
|
+
providerSessionState: summaryOptions.providerSessionState,
|
|
1880
|
+
completeImpl: summaryOptions.completeImpl,
|
|
1881
|
+
telemetry: summaryOptions.telemetry,
|
|
1882
|
+
retry: summaryOneshotRetry(summaryOptions),
|
|
1883
|
+
},
|
|
1884
|
+
);
|
|
1885
|
+
nativeSummary = remote.content;
|
|
1886
|
+
nativeEncryptedContent = remote.encryptedContent;
|
|
1887
|
+
nativeUsedTokens = calculatePromptTokens(remote.usage);
|
|
1888
|
+
usedRemoteCompaction = true;
|
|
1889
|
+
} catch (err) {
|
|
1890
|
+
// A user/session abort is a cancellation, not a remote failure —
|
|
1891
|
+
// swallowing it here would downgrade Esc into "fall back to local
|
|
1892
|
+
// summarization" and keep compaction running on an aborted signal.
|
|
1893
|
+
if (signal?.aborted) throw err;
|
|
1894
|
+
nativeCompactionError = selectNativeCompactionError(nativeCompactionError, err);
|
|
1895
|
+
logger.warn("Anthropic server-side compaction failed", {
|
|
1896
|
+
error: err instanceof Error ? err.message : String(err),
|
|
1897
|
+
model: model.id,
|
|
1898
|
+
provider: model.provider,
|
|
1899
|
+
});
|
|
1900
|
+
}
|
|
1901
|
+
}
|
|
1902
|
+
|
|
1787
1903
|
if (!usedRemoteCompaction && nativeCompactionError !== undefined && !summaryOptions.remoteEndpoint) {
|
|
1788
1904
|
throw new NativeCompactionError(nativeCompactionError);
|
|
1789
1905
|
}
|
|
@@ -1791,7 +1907,11 @@ export async function compact(
|
|
|
1791
1907
|
// Generate summaries (can be parallel if both needed) and merge into one
|
|
1792
1908
|
let summary: string;
|
|
1793
1909
|
|
|
1794
|
-
if (
|
|
1910
|
+
if (nativeSummary !== undefined) {
|
|
1911
|
+
// The API wrote a real summary; it is the entry text. The replayed
|
|
1912
|
+
// block below stays verbatim so it matches the opaque state.
|
|
1913
|
+
summary = nativeSummary;
|
|
1914
|
+
} else if (usedRemoteCompaction) {
|
|
1795
1915
|
// Remote compaction (V2 or V1) already compacted remotely; the durable
|
|
1796
1916
|
// history lives in the provider replay payload (preserveData). Skip local
|
|
1797
1917
|
// summarization so a successful remote compaction never pays for a second,
|
|
@@ -1850,6 +1970,22 @@ export async function compact(
|
|
|
1850
1970
|
// Compute file lists and append to summary
|
|
1851
1971
|
const { readFiles, modifiedFiles } = computeFileLists(fileOps);
|
|
1852
1972
|
summary = upsertFileOperations(summary, readFiles, modifiedFiles, fileOps.read);
|
|
1973
|
+
if (nativeSummary !== undefined) {
|
|
1974
|
+
// The replayed block stays byte-identical to the API's summary so it
|
|
1975
|
+
// matches `encryptedContent`. The harness file lists above travel
|
|
1976
|
+
// separately: the converter replaces the summary message with the
|
|
1977
|
+
// block and skips its text, so they would otherwise be invisible to
|
|
1978
|
+
// this provider. Every other provider keeps reading the entry text.
|
|
1979
|
+
const filesText = upsertFileOperations("", readFiles, modifiedFiles, fileOps.read) || undefined;
|
|
1980
|
+
preserveData = withAnthropicCompactionPreserveData(preserveData, {
|
|
1981
|
+
provider: model.provider,
|
|
1982
|
+
content: nativeSummary,
|
|
1983
|
+
...(nativeEncryptedContent ? { encryptedContent: nativeEncryptedContent } : {}),
|
|
1984
|
+
...(filesText ? { filesText } : {}),
|
|
1985
|
+
model: model.id,
|
|
1986
|
+
usedTokens: nativeUsedTokens,
|
|
1987
|
+
});
|
|
1988
|
+
}
|
|
1853
1989
|
|
|
1854
1990
|
if (!firstKeptEntryId) {
|
|
1855
1991
|
throw new Error("First kept entry has no ID - session may need migration");
|
package/src/compaction/index.ts
CHANGED
package/src/compaction/openai.ts
CHANGED
|
@@ -66,14 +66,14 @@ export * from "./compaction-v2-streaming";
|
|
|
66
66
|
export const OPENAI_REMOTE_COMPACTION_PRESERVE_KEY = "openaiRemoteCompaction";
|
|
67
67
|
|
|
68
68
|
/**
|
|
69
|
-
* Hard ceiling on remote compaction HTTP requests. Unlike every
|
|
70
|
-
* stream (guarded by first-event/idle watchdogs in pi-ai), these are
|
|
71
|
-
* fetches awaiting one non-streamed JSON body — a connection silently
|
|
72
|
-
* by a middlebox would otherwise hang the whole compaction pipeline
|
|
73
|
-
* (frozen "Auto context-full maintenance…" spinner, manual /compact
|
|
74
|
-
* behind it). On timeout the caller falls back to local summarization.
|
|
69
|
+
* Hard ceiling on remote compaction HTTP requests (5 minutes). Unlike every
|
|
70
|
+
* provider stream (guarded by first-event/idle watchdogs in pi-ai), these are
|
|
71
|
+
* raw fetches awaiting one non-streamed JSON body — a connection silently
|
|
72
|
+
* dropped by a middlebox would otherwise hang the whole compaction pipeline
|
|
73
|
+
* forever (frozen "Auto context-full maintenance…" spinner, manual /compact
|
|
74
|
+
* queueing behind it). On timeout the caller falls back to local summarization.
|
|
75
75
|
*/
|
|
76
|
-
export const REMOTE_COMPACTION_TIMEOUT_MS =
|
|
76
|
+
export const REMOTE_COMPACTION_TIMEOUT_MS = 300_000;
|
|
77
77
|
|
|
78
78
|
const DEFAULT_AZURE_API_VERSION = "v1";
|
|
79
79
|
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
{{#if retainedTail}}
|
|
2
|
+
SCOPE: The conversation's final {{#when retainedTail.count "==" 1}}{{retainedTail.role}} message stays{{else}}{{retainedTail.count}} messages, starting with a {{retainedTail.role}} message, stay{{/when}} in context verbatim after your summary. Summarize ONLY the history before those messages. You MUST NOT restate anything from those final messages — the reader sees them right after the summary — and you MUST treat them as the most recent state when describing progress and next steps.
|
|
3
|
+
{{else}}
|
|
4
|
+
SCOPE: The conversation above is the transcript to summarize. The API replaces everything before your summary with it, so nothing you leave out survives into the next context window.
|
|
5
|
+
{{/if}}
|
|
6
|
+
|
|
7
|
+
{{#if extraContext}}
|
|
8
|
+
{{extraContext}}
|
|
9
|
+
|
|
10
|
+
{{/if}}
|
|
11
|
+
{{basePrompt}}
|
|
12
|
+
{{#if customInstructions}}
|
|
13
|
+
|
|
14
|
+
Additional focus: {{customInstructions}}
|
|
15
|
+
{{/if}}
|
|
16
|
+
|
|
17
|
+
You MUST NOT call any tools while writing the summary; respond with the summary text only.
|
package/src/proxy.ts
CHANGED
|
@@ -20,7 +20,6 @@ import {
|
|
|
20
20
|
type StreamingPartialJsonCarrier,
|
|
21
21
|
setStreamingPartialJson,
|
|
22
22
|
} from "@oh-my-pi/pi-ai/utils/block-symbols";
|
|
23
|
-
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
|
|
24
23
|
import { parseStreamingJson, readSseJson } from "@oh-my-pi/pi-utils";
|
|
25
24
|
|
|
26
25
|
// Event stream adapter for proxy SSE events
|
|
@@ -171,7 +170,7 @@ export function streamProxy(model: Model, context: Context, options: ProxyStream
|
|
|
171
170
|
response.body as ReadableStream<Uint8Array>,
|
|
172
171
|
options.signal,
|
|
173
172
|
)) {
|
|
174
|
-
const parsedEvent = processProxyEvent(
|
|
173
|
+
const parsedEvent = processProxyEvent(event, partial, partialJsonByIndex);
|
|
175
174
|
if (parsedEvent) {
|
|
176
175
|
if (parsedEvent.type === "done" || parsedEvent.type === "error") {
|
|
177
176
|
sawTerminalEvent = true;
|
|
@@ -233,7 +232,6 @@ function scrubPartialJson(partial: AssistantMessage): void {
|
|
|
233
232
|
* reads as still-streaming.
|
|
234
233
|
*/
|
|
235
234
|
function processProxyEvent(
|
|
236
|
-
model: Model,
|
|
237
235
|
proxyEvent: ProxyAssistantMessageEvent,
|
|
238
236
|
partial: AssistantMessage,
|
|
239
237
|
partialJsonByIndex: Map<number, string>,
|
|
@@ -375,7 +373,6 @@ function processProxyEvent(
|
|
|
375
373
|
partial.stopReason = proxyEvent.reason;
|
|
376
374
|
partial.usage = proxyEvent.usage;
|
|
377
375
|
if (proxyEvent.content !== undefined) partial.content = proxyEvent.content;
|
|
378
|
-
calculateCost(model, partial.usage);
|
|
379
376
|
scrubPartialJson(partial);
|
|
380
377
|
return { type: "done", reason: proxyEvent.reason, message: partial };
|
|
381
378
|
|
|
@@ -384,7 +381,6 @@ function processProxyEvent(
|
|
|
384
381
|
partial.errorMessage = proxyEvent.errorMessage;
|
|
385
382
|
partial.usage = proxyEvent.usage;
|
|
386
383
|
if (proxyEvent.content !== undefined) partial.content = proxyEvent.content;
|
|
387
|
-
calculateCost(model, partial.usage);
|
|
388
384
|
scrubPartialJson(partial);
|
|
389
385
|
return { type: "error", reason: proxyEvent.reason, error: partial };
|
|
390
386
|
}
|
package/src/tokenizer.ts
CHANGED
|
@@ -219,14 +219,22 @@ export class Tokenizer {
|
|
|
219
219
|
}
|
|
220
220
|
|
|
221
221
|
switch (message.role) {
|
|
222
|
-
case "user":
|
|
223
|
-
|
|
222
|
+
case "user":
|
|
223
|
+
case "developer": {
|
|
224
|
+
// Both roles carry `string | (TextContent | ImageContent)[]` and both are
|
|
225
|
+
// sent to the provider -- convertMessageToLlm handles developer alongside
|
|
226
|
+
// user -- so they are counted alike. Without the developer case the switch
|
|
227
|
+
// fell through to `default: return 0`, and the old annotation narrowed the
|
|
228
|
+
// blocks to text-only, hiding the image charge the toolResult arm applies.
|
|
229
|
+
const content = message.content;
|
|
224
230
|
if (typeof content === "string") {
|
|
225
231
|
fragments.push(content);
|
|
226
232
|
} else if (Array.isArray(content)) {
|
|
227
233
|
for (const block of content) {
|
|
228
234
|
if (block.type === "text" && block.text) {
|
|
229
235
|
fragments.push(block.text);
|
|
236
|
+
} else if (block.type === "image") {
|
|
237
|
+
extra += IMAGE_TOKEN_ESTIMATE;
|
|
230
238
|
}
|
|
231
239
|
}
|
|
232
240
|
}
|