@oh-my-pi/pi-agent-core 18.2.11 → 18.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +24 -0
- package/README.md +3 -2
- package/THIRD-PARTY-NOTICES.txt +3 -3
- package/dist/types/agent-loop.d.ts +6 -0
- package/dist/types/agent.d.ts +14 -0
- package/dist/types/compaction/anthropic.d.ts +22 -52
- package/dist/types/compaction/compaction.d.ts +2 -0
- package/dist/types/compaction/messages.d.ts +9 -0
- package/dist/types/compaction/transcript-tokens.d.ts +14 -1
- package/dist/types/index.d.ts +3 -0
- package/dist/types/live-steering.d.ts +34 -0
- package/dist/types/output-budget.d.ts +43 -0
- package/dist/types/sent-tool-definitions.d.ts +17 -0
- package/dist/types/tool-context.d.ts +31 -0
- package/dist/types/types.d.ts +63 -9
- package/package.json +12 -9
- package/src/agent-loop.ts +219 -44
- package/src/agent.ts +87 -4
- package/src/compaction/anthropic.ts +60 -103
- package/src/compaction/compaction.ts +58 -48
- package/src/compaction/messages.ts +12 -2
- package/src/compaction/prompts/anthropic-compaction-instructions.md +2 -6
- package/src/compaction/transcript-tokens.ts +31 -1
- package/src/index.ts +6 -0
- package/src/live-steering.ts +89 -0
- package/src/output-budget.ts +130 -0
- package/src/sent-tool-definitions.ts +40 -0
- package/src/tool-context.ts +49 -0
- package/src/types.ts +64 -9
package/src/agent.ts
CHANGED
|
@@ -38,7 +38,14 @@ import {
|
|
|
38
38
|
} from "./agent-loop";
|
|
39
39
|
import type { AppendOnlyContextManager } from "./append-only-context";
|
|
40
40
|
import { isProviderRefusalMessage } from "./replay-policy";
|
|
41
|
+
import { SentToolDefinitions } from "./sent-tool-definitions";
|
|
41
42
|
import { Tokenizer, tokenizerEncodingForModel } from "./tokenizer";
|
|
43
|
+
import {
|
|
44
|
+
createAdditionalContextMessage,
|
|
45
|
+
joinAdditionalContext,
|
|
46
|
+
TOOL_RESULT_ADDITIONAL_CONTEXT,
|
|
47
|
+
type ToolResultWithAdditionalContext,
|
|
48
|
+
} from "./tool-context";
|
|
42
49
|
import type {
|
|
43
50
|
AgentBeforeModelCall,
|
|
44
51
|
AgentContext,
|
|
@@ -65,7 +72,7 @@ import { EventLoopKeepalive } from "./utils/yield";
|
|
|
65
72
|
function defaultConvertToLlm(messages: AgentMessage[]): Message[] {
|
|
66
73
|
return messages.filter((m): m is Message => {
|
|
67
74
|
if (m.role === "assistant") return !isProviderRefusalMessage(m);
|
|
68
|
-
return m.role === "user" || m.role === "toolResult";
|
|
75
|
+
return m.role === "user" || m.role === "developer" || m.role === "toolResult";
|
|
69
76
|
});
|
|
70
77
|
}
|
|
71
78
|
|
|
@@ -271,6 +278,11 @@ export interface AgentOptions {
|
|
|
271
278
|
pruneToolDescriptions?: boolean;
|
|
272
279
|
/** Owned tool-calling dialect. Undefined keeps provider-native tool calling. */
|
|
273
280
|
dialect?: Dialect;
|
|
281
|
+
/**
|
|
282
|
+
* Per-request owned-dialect resolver, consulted with the model being requested.
|
|
283
|
+
* Authoritative when set (like {@link serviceTierResolver}): replaces {@link dialect}.
|
|
284
|
+
*/
|
|
285
|
+
dialectResolver?: (model: Model) => Dialect | undefined;
|
|
274
286
|
/**
|
|
275
287
|
* When owned tool calling is active and the model fabricates a tool result
|
|
276
288
|
* mid-turn: `true` (default) aborts the provider request immediately; `false`
|
|
@@ -361,6 +373,12 @@ interface CursorToolResultEntry {
|
|
|
361
373
|
* `message_end` lands in the same chunk as the tool result.
|
|
362
374
|
*/
|
|
363
375
|
pending?: Promise<void>;
|
|
376
|
+
/**
|
|
377
|
+
* Passive context the executor attached via
|
|
378
|
+
* {@link TOOL_RESULT_ADDITIONAL_CONTEXT}, captured before any transformer
|
|
379
|
+
* can replace the message. Injected after the buffered results.
|
|
380
|
+
*/
|
|
381
|
+
additionalContext?: string;
|
|
364
382
|
}
|
|
365
383
|
|
|
366
384
|
type QueuedMessageQueue = "steering" | "followUp";
|
|
@@ -389,6 +407,7 @@ export class Agent {
|
|
|
389
407
|
#convertToLlm: (messages: AgentMessage[]) => Message[] | Promise<Message[]>;
|
|
390
408
|
#transformContext?: (messages: AgentMessage[], signal?: AbortSignal) => Promise<AgentMessage[]>;
|
|
391
409
|
#transformProviderContext?: (context: Context, model: Model) => Context | Promise<Context>;
|
|
410
|
+
#sentToolDefinitions = new SentToolDefinitions();
|
|
392
411
|
#steeringQueue: AgentMessage[] = [];
|
|
393
412
|
#followUpQueue: AgentMessage[] = [];
|
|
394
413
|
#queuedMessageClaims: Partial<Record<QueuedMessageQueue, QueuedMessageClaim>> = {};
|
|
@@ -439,6 +458,7 @@ export class Agent {
|
|
|
439
458
|
#intentTracing: boolean;
|
|
440
459
|
#pruneToolDescriptions: boolean;
|
|
441
460
|
#dialect?: Dialect;
|
|
461
|
+
#dialectResolver?: (model: Model) => Dialect | undefined;
|
|
442
462
|
#abortOnFabricatedToolResult?: boolean;
|
|
443
463
|
#getToolChoice?: () => ToolChoiceDirective | undefined;
|
|
444
464
|
#onToolChoiceUnavailable?: () => void;
|
|
@@ -538,6 +558,7 @@ export class Agent {
|
|
|
538
558
|
this.#intentTracing = opts.intentTracing === true;
|
|
539
559
|
this.#pruneToolDescriptions = opts.pruneToolDescriptions === true;
|
|
540
560
|
this.#dialect = opts.dialect;
|
|
561
|
+
this.#dialectResolver = opts.dialectResolver;
|
|
541
562
|
this.#abortOnFabricatedToolResult = opts.abortOnFabricatedToolResult;
|
|
542
563
|
this.#getToolChoice = opts.getToolChoice;
|
|
543
564
|
this.#onToolChoiceUnavailable = opts.onToolChoiceUnavailable;
|
|
@@ -747,6 +768,33 @@ export class Agent {
|
|
|
747
768
|
this.#hideThinkingSummary = value;
|
|
748
769
|
}
|
|
749
770
|
|
|
771
|
+
/** Strip tool descriptions from provider-bound specs; read per request. */
|
|
772
|
+
get pruneToolDescriptions(): boolean {
|
|
773
|
+
return this.#pruneToolDescriptions;
|
|
774
|
+
}
|
|
775
|
+
|
|
776
|
+
set pruneToolDescriptions(value: boolean) {
|
|
777
|
+
this.#pruneToolDescriptions = value;
|
|
778
|
+
}
|
|
779
|
+
|
|
780
|
+
/** Inject/strip the intent field on tool calls; applies from the next prompt run. */
|
|
781
|
+
get intentTracing(): boolean {
|
|
782
|
+
return this.#intentTracing;
|
|
783
|
+
}
|
|
784
|
+
|
|
785
|
+
set intentTracing(value: boolean) {
|
|
786
|
+
this.#intentTracing = value;
|
|
787
|
+
}
|
|
788
|
+
|
|
789
|
+
/** Abort the provider request on a fabricated tool result; applies from the next prompt run. */
|
|
790
|
+
get abortOnFabricatedToolResult(): boolean | undefined {
|
|
791
|
+
return this.#abortOnFabricatedToolResult;
|
|
792
|
+
}
|
|
793
|
+
|
|
794
|
+
set abortOnFabricatedToolResult(value: boolean | undefined) {
|
|
795
|
+
this.#abortOnFabricatedToolResult = value;
|
|
796
|
+
}
|
|
797
|
+
|
|
750
798
|
/**
|
|
751
799
|
* Get the current max retry delay in milliseconds.
|
|
752
800
|
*/
|
|
@@ -827,7 +875,9 @@ export class Agent {
|
|
|
827
875
|
): Promise<Context> {
|
|
828
876
|
const model = this.#state.model;
|
|
829
877
|
if (!model) throw new Error("No active model on agent");
|
|
830
|
-
const ownedDialect =
|
|
878
|
+
const ownedDialect =
|
|
879
|
+
(this.#dialectResolver ? this.#dialectResolver(model) : this.#dialect) ??
|
|
880
|
+
resolveOwnedDialectFromEnv(Bun.env.PI_DIALECT);
|
|
831
881
|
const messages = normalizeMessagesForProvider(llmMessages, model);
|
|
832
882
|
const tools = ownedDialect
|
|
833
883
|
? []
|
|
@@ -837,6 +887,11 @@ export class Agent {
|
|
|
837
887
|
}) ?? []);
|
|
838
888
|
let context: Context = { systemPrompt, messages, tools };
|
|
839
889
|
if (this.#transformProviderContext) context = await this.#transformProviderContext(context, model);
|
|
890
|
+
// Side requests reuse the main loop's sent definitions without recording their own.
|
|
891
|
+
if (context.tools?.length) {
|
|
892
|
+
const inactiveTools = this.#sentToolDefinitions.inactiveFor(context.messages, context.tools);
|
|
893
|
+
if (inactiveTools) context = { ...context, inactiveTools };
|
|
894
|
+
}
|
|
840
895
|
return context;
|
|
841
896
|
}
|
|
842
897
|
|
|
@@ -1511,7 +1566,10 @@ export class Agent {
|
|
|
1511
1566
|
// that, a transformer resolving after the swap would patch a detached
|
|
1512
1567
|
// object while the persisted result kept the original payload — the
|
|
1513
1568
|
// rewrite silently lost.
|
|
1514
|
-
const entry: CursorToolResultEntry = {
|
|
1569
|
+
const entry: CursorToolResultEntry = {
|
|
1570
|
+
toolResult: message,
|
|
1571
|
+
additionalContext: (message as ToolResultWithAdditionalContext)[TOOL_RESULT_ADDITIONAL_CONTEXT],
|
|
1572
|
+
};
|
|
1515
1573
|
this.#cursorToolResultBuffer.push(entry);
|
|
1516
1574
|
const transform = this.#cursorOnToolResult;
|
|
1517
1575
|
if (transform) {
|
|
@@ -1579,6 +1637,7 @@ export class Agent {
|
|
|
1579
1637
|
preferWebsockets: this.#preferWebsockets,
|
|
1580
1638
|
convertToLlm: this.#convertToLlm,
|
|
1581
1639
|
transformProviderContext: this.#transformProviderContext,
|
|
1640
|
+
sentToolDefinitions: this.#sentToolDefinitions,
|
|
1582
1641
|
transformContext: this.#transformContext,
|
|
1583
1642
|
onPayload: this.#onPayload,
|
|
1584
1643
|
onResponse: this.#onResponse,
|
|
@@ -1616,6 +1675,7 @@ export class Agent {
|
|
|
1616
1675
|
intentTracing: this.#intentTracing,
|
|
1617
1676
|
pruneToolDescriptions: this.#pruneToolDescriptions,
|
|
1618
1677
|
dialect: this.#dialect,
|
|
1678
|
+
getDialect: this.#dialectResolver,
|
|
1619
1679
|
abortOnFabricatedToolResult: this.#abortOnFabricatedToolResult,
|
|
1620
1680
|
appendOnlyContext: this.#appendOnlyContext,
|
|
1621
1681
|
beforeToolCall: this.beforeToolCall ? (ctx, signal) => this.beforeToolCall?.(ctx, signal) : undefined,
|
|
@@ -1773,6 +1833,9 @@ export class Agent {
|
|
|
1773
1833
|
.map(entry => entry.pending);
|
|
1774
1834
|
if (pendingTransforms.length > 0) await Promise.all(pendingTransforms);
|
|
1775
1835
|
const bufferedCursorResults = this.#cursorToolResultBuffer.map(({ toolResult }) => toolResult);
|
|
1836
|
+
const bufferedCursorContext = joinAdditionalContext(
|
|
1837
|
+
this.#cursorToolResultBuffer.map(({ additionalContext }) => additionalContext),
|
|
1838
|
+
);
|
|
1776
1839
|
const retainedToolCallIds = new Set(completedToolCallIds);
|
|
1777
1840
|
for (const { toolCallId } of bufferedCursorResults) retainedToolCallIds.add(toolCallId);
|
|
1778
1841
|
const errorMsg: AssistantMessage =
|
|
@@ -1849,9 +1912,13 @@ export class Agent {
|
|
|
1849
1912
|
this.#emit({ type: "message_end", message: toolResult });
|
|
1850
1913
|
toolResults.push(toolResult);
|
|
1851
1914
|
}
|
|
1915
|
+
const agentEndMessages: AgentMessage[] = [errorMsg, ...toolResults];
|
|
1916
|
+
if (bufferedCursorContext !== undefined) {
|
|
1917
|
+
agentEndMessages.push(this.#emitCursorAdditionalContext(bufferedCursorContext));
|
|
1918
|
+
}
|
|
1852
1919
|
this.#emit({ type: "turn_end", message: errorMsg, toolResults });
|
|
1853
1920
|
turnOpen = false;
|
|
1854
|
-
this.#emit({ type: "agent_end", messages:
|
|
1921
|
+
this.#emit({ type: "agent_end", messages: agentEndMessages });
|
|
1855
1922
|
} else {
|
|
1856
1923
|
this.appendMessage(errorMsg);
|
|
1857
1924
|
this.#state.error = errorMessage;
|
|
@@ -1923,8 +1990,24 @@ export class Agent {
|
|
|
1923
1990
|
this.appendMessage(toolResult);
|
|
1924
1991
|
this.#emit({ type: "message_end", message: toolResult });
|
|
1925
1992
|
}
|
|
1993
|
+
const additionalContext = joinAdditionalContext(buffer.map(entry => entry.additionalContext));
|
|
1994
|
+
if (additionalContext !== undefined) this.#emitCursorAdditionalContext(additionalContext);
|
|
1926
1995
|
} finally {
|
|
1927
1996
|
this.#cursorToolResultDrain = undefined;
|
|
1928
1997
|
}
|
|
1929
1998
|
}
|
|
1999
|
+
|
|
2000
|
+
/**
|
|
2001
|
+
* Append passive context reported by Cursor exec-channel tools after their
|
|
2002
|
+
* results, mirroring the loop's post-batch developer message. Cursor runs
|
|
2003
|
+
* those tools server-side mid-stream, so the context reaches the next
|
|
2004
|
+
* provider request instead of the current one.
|
|
2005
|
+
*/
|
|
2006
|
+
#emitCursorAdditionalContext(text: string): AgentMessage {
|
|
2007
|
+
const message = createAdditionalContextMessage(text);
|
|
2008
|
+
this.#emit({ type: "message_start", message });
|
|
2009
|
+
this.appendMessage(message);
|
|
2010
|
+
this.#emit({ type: "message_end", message });
|
|
2011
|
+
return message;
|
|
2012
|
+
}
|
|
1930
2013
|
}
|
|
@@ -1,21 +1,14 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Anthropic
|
|
2
|
+
* Anthropic on-demand compaction (`compact-2026-09-04` beta).
|
|
3
3
|
*
|
|
4
|
-
* The
|
|
5
|
-
* prompt, tools
|
|
6
|
-
*
|
|
7
|
-
* cached prefix and stops; the summary arrives as a `compaction` block that
|
|
8
|
-
* the provider surfaces as an `anthropicCompaction` payload. The summary is
|
|
9
|
-
* plain text, so it doubles as the compaction entry's readable summary for
|
|
10
|
-
* every other provider, while the Anthropic provider replays it as a native
|
|
11
|
-
* block (the API drops everything that precedes it). The retained tail after
|
|
12
|
-
* the cut point is replayed from session entries exactly like a local summary.
|
|
4
|
+
* The request sends only the prefix to summarize, with the live conversation's
|
|
5
|
+
* system prompt, tools and thinking settings. The returned signed block
|
|
6
|
+
* replaces that prefix; the retained tail is replayed from session entries.
|
|
13
7
|
*/
|
|
14
8
|
|
|
15
9
|
import type {
|
|
16
10
|
AnthropicCompactionPayload,
|
|
17
11
|
ApiKey,
|
|
18
|
-
AssistantMessage,
|
|
19
12
|
Effort,
|
|
20
13
|
Message,
|
|
21
14
|
Model,
|
|
@@ -27,26 +20,18 @@ import * as AIError from "@oh-my-pi/pi-ai/error";
|
|
|
27
20
|
import { supportsAnthropicCompaction } from "@oh-my-pi/pi-ai/providers/anthropic-compaction";
|
|
28
21
|
import { isRecord, prompt } from "@oh-my-pi/pi-utils";
|
|
29
22
|
import { type InstrumentedChatSpanOptions, instrumentedCompleteSimple } from "../telemetry";
|
|
23
|
+
import type { AgentMessage } from "../types";
|
|
30
24
|
import anthropicCompactionInstructionsPrompt from "./prompts/anthropic-compaction-instructions.md" with { type: "text" };
|
|
31
25
|
|
|
32
26
|
export const ANTHROPIC_COMPACTION_PRESERVE_KEY = "anthropicCompaction";
|
|
33
27
|
|
|
34
|
-
/** The API rejects a `compact_20260112` trigger below this many input tokens. */
|
|
35
|
-
export const ANTHROPIC_COMPACTION_MIN_TRIGGER_TOKENS = 50_000;
|
|
36
|
-
|
|
37
|
-
/**
|
|
38
|
-
* Smallest context the native lane accepts. The trigger sits at the API
|
|
39
|
-
* floor, so a prompt that lands below it is answered instead of compacted;
|
|
40
|
-
* the margin over the floor absorbs the difference between the last reported
|
|
41
|
-
* context size and the compaction request's own input.
|
|
42
|
-
*/
|
|
43
|
-
export const ANTHROPIC_COMPACTION_MIN_CONTEXT_TOKENS = 55_000;
|
|
44
|
-
|
|
45
28
|
/** Summary persisted under {@link ANTHROPIC_COMPACTION_PRESERVE_KEY}. */
|
|
46
29
|
export interface AnthropicCompactionPreserveData {
|
|
47
30
|
provider: string;
|
|
48
31
|
content: string;
|
|
49
|
-
/**
|
|
32
|
+
/** Signature attached to an on-demand block; replayed verbatim. */
|
|
33
|
+
signature?: string;
|
|
34
|
+
/** Legacy threshold block state; replay-only. */
|
|
50
35
|
encryptedContent?: string;
|
|
51
36
|
/** Harness file metadata (`<files>` section) replayed after the native block. */
|
|
52
37
|
filesText?: string;
|
|
@@ -82,6 +67,9 @@ export function getPreservedAnthropicCompactionData(
|
|
|
82
67
|
return {
|
|
83
68
|
provider: candidate.provider,
|
|
84
69
|
content: candidate.content,
|
|
70
|
+
...(typeof candidate.signature === "string" && candidate.signature.length > 0
|
|
71
|
+
? { signature: candidate.signature }
|
|
72
|
+
: {}),
|
|
85
73
|
...(typeof candidate.encryptedContent === "string" && candidate.encryptedContent.length > 0
|
|
86
74
|
? { encryptedContent: candidate.encryptedContent }
|
|
87
75
|
: {}),
|
|
@@ -118,93 +106,63 @@ export function getAnthropicCompactionPayload(
|
|
|
118
106
|
type: "anthropicCompaction",
|
|
119
107
|
provider: preserved.provider,
|
|
120
108
|
content: preserved.content,
|
|
109
|
+
...(preserved.signature ? { signature: preserved.signature } : {}),
|
|
121
110
|
...(preserved.encryptedContent ? { encryptedContent: preserved.encryptedContent } : {}),
|
|
122
111
|
...(preserved.filesText ? { filesText: preserved.filesText } : {}),
|
|
123
112
|
};
|
|
124
113
|
}
|
|
125
114
|
|
|
126
115
|
/**
|
|
127
|
-
*
|
|
128
|
-
*
|
|
129
|
-
*
|
|
130
|
-
* the whole conversation so the prompt cache the live turn wrote is read, but
|
|
131
|
-
* the summary must cover only the history before that tail — the local
|
|
132
|
-
* summarizer never sees the tail, and the rebuilt context replays it after the
|
|
133
|
-
* summary. Counting mirrors the provider's message conversion (consecutive
|
|
134
|
-
* tool results collapse into one user message; developer messages are user
|
|
135
|
-
* messages). Structured for the prompt template, which renders the
|
|
136
|
-
* singular/plural wording; the description quotes no content: quoting the
|
|
137
|
-
* tail would hand the summarizer the very facts it must leave to the tail.
|
|
116
|
+
* Move the existing keep-tail boundary forward until the summary/tail boundary
|
|
117
|
+
* alternates wire roles and no tool call is separated from its result. If no
|
|
118
|
+
* boundary is safe, the request summarizes the entire snapshot (empty tail).
|
|
138
119
|
*/
|
|
139
|
-
export
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
const isToolResult = message.role === "toolResult";
|
|
151
|
-
if (!(isToolResult && previousWasToolResult)) count += 1;
|
|
152
|
-
previousWasToolResult = isToolResult;
|
|
120
|
+
export function findAnthropicCompactionCut(
|
|
121
|
+
messages: readonly (AgentMessage | { role: "system"; content: string; timestamp: number })[],
|
|
122
|
+
initialCut: number,
|
|
123
|
+
): number {
|
|
124
|
+
let calls: Map<string, number> | undefined;
|
|
125
|
+
for (let i = 0; i < messages.length; i++) {
|
|
126
|
+
const message = messages[i];
|
|
127
|
+
if (message.role !== "assistant") continue;
|
|
128
|
+
for (const block of message.content) {
|
|
129
|
+
if (block.type === "toolCall") (calls ??= new Map()).set(block.id, i);
|
|
130
|
+
}
|
|
153
131
|
}
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
count += 1;
|
|
132
|
+
let lastResult: Int32Array | undefined;
|
|
133
|
+
if (calls) {
|
|
134
|
+
lastResult = new Int32Array(messages.length);
|
|
135
|
+
for (let i = 0; i < messages.length; i++) {
|
|
136
|
+
const message = messages[i];
|
|
137
|
+
if (message.role !== "toolResult") continue;
|
|
138
|
+
const callIndex = calls.get(message.toolCallId);
|
|
139
|
+
if (callIndex !== undefined) lastResult[callIndex] = i;
|
|
140
|
+
}
|
|
164
141
|
}
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
switch (block.type) {
|
|
176
|
-
case "text":
|
|
177
|
-
return block.text.trim().length > 0;
|
|
178
|
-
case "toolCall":
|
|
179
|
-
case "anthropicServerTool":
|
|
180
|
-
return true;
|
|
181
|
-
case "thinking":
|
|
182
|
-
return block.thinking.trim().length > 0 || (block.thinkingSignature ?? "").trim().length > 0;
|
|
183
|
-
default:
|
|
184
|
-
return false;
|
|
142
|
+
let protectedThrough = -1;
|
|
143
|
+
for (let i = 0; i < initialCut; i++) protectedThrough = Math.max(protectedThrough, lastResult?.[i] ?? -1);
|
|
144
|
+
for (let cut = initialCut; cut < messages.length; cut++) {
|
|
145
|
+
protectedThrough = Math.max(protectedThrough, lastResult?.[cut - 1] ?? -1);
|
|
146
|
+
if (cut <= protectedThrough) continue;
|
|
147
|
+
const first = messages[cut];
|
|
148
|
+
if (first.role === "system" || first.role === "developer") continue;
|
|
149
|
+
const previous = messages[cut - 1];
|
|
150
|
+
if (!previous) continue;
|
|
151
|
+
if ((previous.role === "assistant") !== (first.role === "assistant")) return cut;
|
|
185
152
|
}
|
|
153
|
+
return messages.length;
|
|
186
154
|
}
|
|
187
155
|
|
|
188
|
-
/**
|
|
189
|
-
* Summarization prompt sent as the edit's `instructions`, which replace the
|
|
190
|
-
* API default entirely. The template lays out the retained-tail boundary
|
|
191
|
-
* first, so the summary covers only the history the rebuilt context drops,
|
|
192
|
-
* then the caller's extra context, the same structure prompt as the local
|
|
193
|
-
* summarizer, the caller's focus, and the tool-abstention clause the API
|
|
194
|
-
* recommends when tools are defined (a summarization pass that calls a tool
|
|
195
|
-
* yields no summary).
|
|
196
|
-
*/
|
|
156
|
+
/** Instructions replace the API default; the request contains only summarized messages. */
|
|
197
157
|
export function buildAnthropicCompactionInstructions(
|
|
198
158
|
basePrompt: string,
|
|
199
159
|
customInstructions: string | undefined,
|
|
200
160
|
extraContext: string | undefined,
|
|
201
|
-
retainedTail: RetainedTailScope | undefined,
|
|
202
161
|
): string {
|
|
203
162
|
return prompt.render(anthropicCompactionInstructionsPrompt, {
|
|
204
163
|
basePrompt,
|
|
205
164
|
customInstructions,
|
|
206
165
|
extraContext,
|
|
207
|
-
retainedTail,
|
|
208
166
|
});
|
|
209
167
|
}
|
|
210
168
|
|
|
@@ -219,7 +177,7 @@ export interface AnthropicNativeCompactionRequest {
|
|
|
219
177
|
|
|
220
178
|
export interface AnthropicNativeCompactionResponse {
|
|
221
179
|
content: string;
|
|
222
|
-
|
|
180
|
+
signature: string;
|
|
223
181
|
usage: Usage;
|
|
224
182
|
model: string;
|
|
225
183
|
}
|
|
@@ -239,15 +197,13 @@ export interface AnthropicNativeCompactionOptions
|
|
|
239
197
|
Pick<InstrumentedChatSpanOptions, "completeImpl" | "telemetry" | "retry"> {}
|
|
240
198
|
|
|
241
199
|
/**
|
|
242
|
-
* Run one compaction request and return the summary the API wrote
|
|
243
|
-
* opaque `encrypted_content` the API attached for the replay. `completeSimple`
|
|
200
|
+
* Run one compaction request and return the summary and signature the API wrote. `completeSimple`
|
|
244
201
|
* resolves terminal failures as messages, so their classification is restored
|
|
245
202
|
* here: an aborted response is an `AbortError` (a cancellation, never a native
|
|
246
203
|
* failure) and an error response keeps its HTTP status, so auth and timeout
|
|
247
204
|
* handling downstream classify it the same way as the OpenAI lanes. A response
|
|
248
|
-
* without a summary is a native failure
|
|
249
|
-
*
|
|
250
|
-
* the model called a tool during summarization.
|
|
205
|
+
* without a summary is a native failure, including tool use, refusals and
|
|
206
|
+
* output limits; the configured method order can then choose a fallback.
|
|
251
207
|
*/
|
|
252
208
|
export async function requestAnthropicNativeCompaction(
|
|
253
209
|
model: Model<"anthropic-messages">,
|
|
@@ -271,11 +227,7 @@ export async function requestAnthropicNativeCompaction(
|
|
|
271
227
|
promptCacheKey: options.promptCacheKey,
|
|
272
228
|
providerSessionState: options.providerSessionState,
|
|
273
229
|
maxInFlightRequests: options.maxInFlightRequests,
|
|
274
|
-
anthropicCompaction: {
|
|
275
|
-
triggerInputTokens: ANTHROPIC_COMPACTION_MIN_TRIGGER_TOKENS,
|
|
276
|
-
pauseAfterCompaction: true,
|
|
277
|
-
instructions: request.instructions,
|
|
278
|
-
},
|
|
230
|
+
anthropicCompaction: { instructions: request.instructions },
|
|
279
231
|
},
|
|
280
232
|
{
|
|
281
233
|
telemetry: options.telemetry,
|
|
@@ -294,16 +246,21 @@ export async function requestAnthropicNativeCompaction(
|
|
|
294
246
|
: new AIError.ProviderHttpError(message, response.errorStatus);
|
|
295
247
|
}
|
|
296
248
|
const payload = response.providerPayload;
|
|
297
|
-
if (
|
|
249
|
+
if (
|
|
250
|
+
response.stopDetails?.type !== "compaction" ||
|
|
251
|
+
payload?.type !== "anthropicCompaction" ||
|
|
252
|
+
!payload.content ||
|
|
253
|
+
!payload.signature
|
|
254
|
+
) {
|
|
298
255
|
throw new Error(
|
|
299
256
|
response.stopDetails?.type === "compaction"
|
|
300
|
-
? "Anthropic compaction returned no summary"
|
|
257
|
+
? "Anthropic compaction returned no signed summary"
|
|
301
258
|
: "Anthropic compaction response carried no compaction block",
|
|
302
259
|
);
|
|
303
260
|
}
|
|
304
261
|
return {
|
|
305
262
|
content: payload.content,
|
|
306
|
-
|
|
263
|
+
signature: payload.signature,
|
|
307
264
|
usage: response.usage,
|
|
308
265
|
model: response.model,
|
|
309
266
|
};
|