@oh-my-pi/pi-agent-core 18.2.11 → 18.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +24 -0
- package/README.md +3 -2
- package/THIRD-PARTY-NOTICES.txt +3 -3
- package/dist/types/agent-loop.d.ts +6 -0
- package/dist/types/agent.d.ts +14 -0
- package/dist/types/compaction/anthropic.d.ts +22 -52
- package/dist/types/compaction/compaction.d.ts +2 -0
- package/dist/types/compaction/messages.d.ts +9 -0
- package/dist/types/compaction/transcript-tokens.d.ts +14 -1
- package/dist/types/index.d.ts +3 -0
- package/dist/types/live-steering.d.ts +34 -0
- package/dist/types/output-budget.d.ts +43 -0
- package/dist/types/sent-tool-definitions.d.ts +17 -0
- package/dist/types/tool-context.d.ts +31 -0
- package/dist/types/types.d.ts +63 -9
- package/package.json +12 -9
- package/src/agent-loop.ts +219 -44
- package/src/agent.ts +87 -4
- package/src/compaction/anthropic.ts +60 -103
- package/src/compaction/compaction.ts +58 -48
- package/src/compaction/messages.ts +12 -2
- package/src/compaction/prompts/anthropic-compaction-instructions.md +2 -6
- package/src/compaction/transcript-tokens.ts +31 -1
- package/src/index.ts +6 -0
- package/src/live-steering.ts +89 -0
- package/src/output-budget.ts +130 -0
- package/src/sent-tool-definitions.ts +40 -0
- package/src/tool-context.ts +49 -0
- package/src/types.ts +64 -9
|
@@ -43,9 +43,8 @@ import { ThinkingLevel } from "../thinking";
|
|
|
43
43
|
import { Tokenizer } from "../tokenizer";
|
|
44
44
|
import type { AgentMessage } from "../types";
|
|
45
45
|
import {
|
|
46
|
-
ANTHROPIC_COMPACTION_MIN_CONTEXT_TOKENS,
|
|
47
46
|
buildAnthropicCompactionInstructions,
|
|
48
|
-
|
|
47
|
+
findAnthropicCompactionCut,
|
|
49
48
|
getPreservedAnthropicCompactionData,
|
|
50
49
|
requestAnthropicNativeCompaction,
|
|
51
50
|
shouldUseAnthropicNativeCompaction,
|
|
@@ -1252,6 +1251,8 @@ export interface CompactionPreparation {
|
|
|
1252
1251
|
turnPrefixMessages: AgentMessage[];
|
|
1253
1252
|
/** Messages kept in full after compaction (recent history) */
|
|
1254
1253
|
recentMessages: AgentMessage[];
|
|
1254
|
+
/** Entry IDs parallel to recentMessages, for an Anthropic-safe keep-tail boundary. */
|
|
1255
|
+
recentEntryIds?: string[];
|
|
1255
1256
|
/** Whether this is a split turn (cut point in middle of turn) */
|
|
1256
1257
|
isSplitTurn: boolean;
|
|
1257
1258
|
tokensBefore: number;
|
|
@@ -1380,16 +1381,22 @@ export function prepareCompaction(
|
|
|
1380
1381
|
}
|
|
1381
1382
|
}
|
|
1382
1383
|
} else if (previousCompaction) {
|
|
1383
|
-
// Local summaries exclude the retained tail, whose
|
|
1384
|
-
// the compaction record.
|
|
1385
|
-
// backwards: advisor snapshots put all retained messages after the summary
|
|
1386
|
-
// and may carry a keep ID from their previous, differently indexed snapshot.
|
|
1384
|
+
// Local and Anthropic summaries exclude the retained tail, whose
|
|
1385
|
+
// original entries precede the compaction record.
|
|
1387
1386
|
for (let i = resetBoundaryIndex + 1; i < prevCompactionIndex; i++) {
|
|
1388
1387
|
if (pathEntries[i].id === previousCompaction.firstKeptEntryId) {
|
|
1389
1388
|
boundaryStart = i;
|
|
1390
1389
|
break;
|
|
1391
1390
|
}
|
|
1392
1391
|
}
|
|
1392
|
+
if (previousCompaction.firstKeptEntryId === "" && previousCompaction.providerReplayThroughEntryId) {
|
|
1393
|
+
// An empty snapshot tail can still have turns appended during
|
|
1394
|
+
// background compaction; those start after the summarized snapshot.
|
|
1395
|
+
const snapshotIdx = pathEntries.findIndex(
|
|
1396
|
+
entry => entry.id === previousCompaction.providerReplayThroughEntryId,
|
|
1397
|
+
);
|
|
1398
|
+
if (snapshotIdx >= 0 && snapshotIdx < prevCompactionIndex) boundaryStart = snapshotIdx + 1;
|
|
1399
|
+
}
|
|
1393
1400
|
}
|
|
1394
1401
|
|
|
1395
1402
|
// Keep original IDs beside the converted messages so estimation, cutting,
|
|
@@ -1437,6 +1444,11 @@ export function prepareCompaction(
|
|
|
1437
1444
|
return undefined;
|
|
1438
1445
|
}
|
|
1439
1446
|
|
|
1447
|
+
const recentEntryIds: string[] = [];
|
|
1448
|
+
for (let i = cutPoint.firstKeptEntryIndex; i < compactionEntries.length; i++) {
|
|
1449
|
+
recentEntryIds.push(compactionEntries[i].id);
|
|
1450
|
+
}
|
|
1451
|
+
|
|
1440
1452
|
// Extract file operations from messages and previous compaction
|
|
1441
1453
|
const fileOps = extractFileOperations(messagesToSummarize, pathEntries, prevCompactionIndex);
|
|
1442
1454
|
|
|
@@ -1452,6 +1464,7 @@ export function prepareCompaction(
|
|
|
1452
1464
|
messagesToSummarize,
|
|
1453
1465
|
turnPrefixMessages,
|
|
1454
1466
|
recentMessages,
|
|
1467
|
+
recentEntryIds,
|
|
1455
1468
|
isSplitTurn: cutPoint.isSplitTurn,
|
|
1456
1469
|
tokensBefore,
|
|
1457
1470
|
previousSummary: previousCompaction?.summary,
|
|
@@ -1579,6 +1592,7 @@ export async function compact(
|
|
|
1579
1592
|
messagesToSummarize,
|
|
1580
1593
|
turnPrefixMessages,
|
|
1581
1594
|
recentMessages,
|
|
1595
|
+
recentEntryIds,
|
|
1582
1596
|
isSplitTurn,
|
|
1583
1597
|
tokensBefore,
|
|
1584
1598
|
previousSummary,
|
|
@@ -1826,35 +1840,19 @@ export async function compact(
|
|
|
1826
1840
|
}
|
|
1827
1841
|
}
|
|
1828
1842
|
|
|
1829
|
-
//
|
|
1830
|
-
//
|
|
1831
|
-
// real text, persisted both as the entry summary and as the native replay
|
|
1832
|
-
// payload. A context below the API's trigger floor cannot compact remotely
|
|
1833
|
-
// and takes the local summarizer instead — an eligibility boundary, not a
|
|
1834
|
-
// failure.
|
|
1843
|
+
// On-demand compaction summarizes only the prefix; the tail is never sent
|
|
1844
|
+
// to this request and is replayed after the returned signed block.
|
|
1835
1845
|
let nativeSummary: string | undefined;
|
|
1836
|
-
let
|
|
1846
|
+
let nativeSignature: string | undefined;
|
|
1847
|
+
let nativeFirstKeptEntryId = firstKeptEntryId;
|
|
1837
1848
|
let nativeUsedTokens: number | undefined;
|
|
1838
|
-
if (
|
|
1839
|
-
!usedRemoteCompaction &&
|
|
1840
|
-
settings.remoteEnabled !== false &&
|
|
1841
|
-
shouldUseAnthropicNativeCompaction(model) &&
|
|
1842
|
-
tokensBefore >= ANTHROPIC_COMPACTION_MIN_CONTEXT_TOKENS
|
|
1843
|
-
) {
|
|
1849
|
+
if (!usedRemoteCompaction && settings.remoteEnabled !== false && shouldUseAnthropicNativeCompaction(model)) {
|
|
1844
1850
|
const previousNative = getPreservedAnthropicCompactionData(previousPreserveData);
|
|
1845
|
-
// Lead with the previous summary
|
|
1846
|
-
// natively when this provider wrote it, as text otherwise.
|
|
1847
|
-
//
|
|
1848
|
-
//
|
|
1849
|
-
//
|
|
1850
|
-
// The rewrite marker must precede every message this request replays —
|
|
1851
|
-
// summarized history and retained tail alike — exactly like the live
|
|
1852
|
-
// context rebuild predates its tail. A previous compaction's commit
|
|
1853
|
-
// timestamp is newer than re-retained or re-summarized turns, so
|
|
1854
|
-
// reusing it would strip their bound thinking only in this request,
|
|
1855
|
-
// diverging from the cached live prefix (and possibly dropping below
|
|
1856
|
-
// the trigger). Manually built preparations with no input at all fall
|
|
1857
|
-
// back to the current time.
|
|
1851
|
+
// Lead with the previous summary as the live context renders it:
|
|
1852
|
+
// natively when this provider wrote it, as text otherwise.
|
|
1853
|
+
// A prior snapcompact archive is already merged into that summary
|
|
1854
|
+
// text. Predate the rewrite marker before all replayed messages so
|
|
1855
|
+
// retained thinking stays bound to the original prefix.
|
|
1858
1856
|
const firstReplayed = messagesToSummarize[0] ?? turnPrefixMessages[0] ?? recentMessages[0];
|
|
1859
1857
|
const previousSummaryAt =
|
|
1860
1858
|
firstReplayed !== undefined ? new Date(firstReplayed.timestamp - 1).toISOString() : new Date().toISOString();
|
|
@@ -1866,6 +1864,7 @@ export async function compact(
|
|
|
1866
1864
|
type: "anthropicCompaction",
|
|
1867
1865
|
provider: previousNative.provider,
|
|
1868
1866
|
content: previousNative.content,
|
|
1867
|
+
...(previousNative.signature ? { signature: previousNative.signature } : {}),
|
|
1869
1868
|
...(previousNative.encryptedContent
|
|
1870
1869
|
? { encryptedContent: previousNative.encryptedContent }
|
|
1871
1870
|
: {}),
|
|
@@ -1874,29 +1873,40 @@ export async function compact(
|
|
|
1874
1873
|
: undefined,
|
|
1875
1874
|
})
|
|
1876
1875
|
: undefined;
|
|
1877
|
-
const
|
|
1878
|
-
const
|
|
1879
|
-
const
|
|
1880
|
-
|
|
1881
|
-
|
|
1882
|
-
|
|
1883
|
-
|
|
1884
|
-
|
|
1885
|
-
|
|
1886
|
-
|
|
1876
|
+
const allMessages = [...messagesToSummarize, ...turnPrefixMessages, ...recentMessages];
|
|
1877
|
+
const originalCut = messagesToSummarize.length + turnPrefixMessages.length;
|
|
1878
|
+
const nativeCut = findAnthropicCompactionCut(allMessages, originalCut);
|
|
1879
|
+
// Hand-built preparations lacking entry IDs cannot move their persisted
|
|
1880
|
+
// boundary into the tail. Summarizing all is still safe.
|
|
1881
|
+
const safeCut =
|
|
1882
|
+
nativeCut > originalCut && nativeCut < allMessages.length && !recentEntryIds?.[nativeCut - originalCut]
|
|
1883
|
+
? allMessages.length
|
|
1884
|
+
: nativeCut;
|
|
1885
|
+
nativeFirstKeptEntryId =
|
|
1886
|
+
safeCut === allMessages.length
|
|
1887
|
+
? ""
|
|
1888
|
+
: safeCut === originalCut
|
|
1889
|
+
? firstKeptEntryId
|
|
1890
|
+
: (recentEntryIds?.[safeCut - originalCut] ?? "");
|
|
1891
|
+
for (let i = originalCut; i < safeCut; i++) extractFileOpsFromMessage(allMessages[i], fileOps);
|
|
1892
|
+
const messages = (summaryOptions.convertToLlm ?? defaultConvertToLlm)([
|
|
1893
|
+
...(previousSummaryMessage ? [previousSummaryMessage] : []),
|
|
1894
|
+
...allMessages.slice(0, safeCut),
|
|
1895
|
+
]);
|
|
1887
1896
|
try {
|
|
1888
1897
|
const remote = await requestAnthropicNativeCompaction(
|
|
1889
1898
|
model,
|
|
1890
1899
|
apiKey,
|
|
1891
1900
|
{
|
|
1892
|
-
|
|
1901
|
+
// The live prompt, not the local summarizer's synthetic system
|
|
1902
|
+
// prompt: kept thinking remains valid only under identical controls.
|
|
1903
|
+
systemPrompt: summaryOptions.remoteSystemPrompt ?? [],
|
|
1893
1904
|
messages,
|
|
1894
1905
|
tools: summaryOptions.tools,
|
|
1895
1906
|
instructions: buildAnthropicCompactionInstructions(
|
|
1896
1907
|
summaryOptions.promptOverride ?? SUMMARIZATION_PROMPT,
|
|
1897
1908
|
customInstructions,
|
|
1898
1909
|
formatAdditionalContext(summaryOptions.extraContext).trim() || undefined,
|
|
1899
|
-
describeRetainedTail(retainedTail),
|
|
1900
1910
|
),
|
|
1901
1911
|
maxTokens: Math.min(Math.floor(0.8 * reserveTokens), MAX_SUMMARY_TOKENS),
|
|
1902
1912
|
reasoning: resolveCompactionEffort(model, summaryOptions.thinkingLevel),
|
|
@@ -1915,7 +1925,7 @@ export async function compact(
|
|
|
1915
1925
|
},
|
|
1916
1926
|
);
|
|
1917
1927
|
nativeSummary = remote.content;
|
|
1918
|
-
|
|
1928
|
+
nativeSignature = remote.signature;
|
|
1919
1929
|
nativeUsedTokens = calculatePromptTokens(remote.usage);
|
|
1920
1930
|
usedRemoteCompaction = true;
|
|
1921
1931
|
} catch (err) {
|
|
@@ -2004,7 +2014,7 @@ export async function compact(
|
|
|
2004
2014
|
summary = upsertFileOperations(summary, readFiles, modifiedFiles, fileOps.read);
|
|
2005
2015
|
if (nativeSummary !== undefined) {
|
|
2006
2016
|
// The replayed block stays byte-identical to the API's summary so it
|
|
2007
|
-
// matches
|
|
2017
|
+
// matches its signature. The harness file lists above travel
|
|
2008
2018
|
// separately: the converter replaces the summary message with the
|
|
2009
2019
|
// block and skips its text, so they would otherwise be invisible to
|
|
2010
2020
|
// this provider. Every other provider keeps reading the entry text.
|
|
@@ -2012,7 +2022,7 @@ export async function compact(
|
|
|
2012
2022
|
preserveData = withAnthropicCompactionPreserveData(preserveData, {
|
|
2013
2023
|
provider: model.provider,
|
|
2014
2024
|
content: nativeSummary,
|
|
2015
|
-
...(
|
|
2025
|
+
...(nativeSignature ? { signature: nativeSignature } : {}),
|
|
2016
2026
|
...(filesText ? { filesText } : {}),
|
|
2017
2027
|
model: model.id,
|
|
2018
2028
|
usedTokens: nativeUsedTokens,
|
|
@@ -2034,7 +2044,7 @@ export async function compact(
|
|
|
2034
2044
|
return {
|
|
2035
2045
|
summary,
|
|
2036
2046
|
shortSummary,
|
|
2037
|
-
firstKeptEntryId,
|
|
2047
|
+
firstKeptEntryId: nativeSummary !== undefined ? nativeFirstKeptEntryId : firstKeptEntryId,
|
|
2038
2048
|
tokensBefore,
|
|
2039
2049
|
details: { readFiles, modifiedFiles } as CompactionDetails,
|
|
2040
2050
|
preserveData: finalPreserveData,
|
|
@@ -64,6 +64,13 @@ export interface CompactionSummaryMessage {
|
|
|
64
64
|
images?: ImageContent[];
|
|
65
65
|
/** Post-pass dead-end warning attached to this compaction (progress guard). */
|
|
66
66
|
warning?: string;
|
|
67
|
+
/**
|
|
68
|
+
* Thinking-binding rewrite marker when it must differ from `timestamp`: a
|
|
69
|
+
* natively replayed summary predates it before the retained tail so that
|
|
70
|
+
* tail's bound thinking stays valid. `timestamp` remains the commit time,
|
|
71
|
+
* which is what invalidates the tail's pre-compaction usage reports.
|
|
72
|
+
*/
|
|
73
|
+
historyRewriteAt?: number;
|
|
67
74
|
timestamp: number;
|
|
68
75
|
}
|
|
69
76
|
|
|
@@ -135,6 +142,8 @@ export interface CompactionSummaryMessageOptions {
|
|
|
135
142
|
method?: string;
|
|
136
143
|
/** Estimated context tokens after the rewrite, for display alongside `tokensBefore`. */
|
|
137
144
|
tokensAfter?: number;
|
|
145
|
+
/** See {@link CompactionSummaryMessage.historyRewriteAt}. */
|
|
146
|
+
historyRewriteAt?: number;
|
|
138
147
|
}
|
|
139
148
|
|
|
140
149
|
export function createCompactionSummaryMessage(
|
|
@@ -143,7 +152,7 @@ export function createCompactionSummaryMessage(
|
|
|
143
152
|
timestamp: string,
|
|
144
153
|
options: CompactionSummaryMessageOptions = {},
|
|
145
154
|
): CompactionSummaryMessage {
|
|
146
|
-
const { shortSummary, providerPayload, images, blocks, warning, method, tokensAfter } = options;
|
|
155
|
+
const { shortSummary, providerPayload, images, blocks, warning, method, tokensAfter, historyRewriteAt } = options;
|
|
147
156
|
const imageBlocks =
|
|
148
157
|
blocks?.filter((block): block is ImageContent => block.type === "image") ??
|
|
149
158
|
(images && images.length > 0 ? images : undefined);
|
|
@@ -158,6 +167,7 @@ export function createCompactionSummaryMessage(
|
|
|
158
167
|
blocks: blocks && blocks.length > 0 ? blocks : undefined,
|
|
159
168
|
images: imageBlocks && imageBlocks.length > 0 ? imageBlocks : undefined,
|
|
160
169
|
warning,
|
|
170
|
+
historyRewriteAt,
|
|
161
171
|
timestamp: new Date(timestamp).getTime(),
|
|
162
172
|
};
|
|
163
173
|
}
|
|
@@ -246,7 +256,7 @@ export function convertMessageToLlm(message: AgentMessage): Message | undefined
|
|
|
246
256
|
...(message.images ?? []),
|
|
247
257
|
],
|
|
248
258
|
attribution: "agent",
|
|
249
|
-
historyRewriteAt: message.timestamp,
|
|
259
|
+
historyRewriteAt: message.historyRewriteAt ?? message.timestamp,
|
|
250
260
|
providerPayload: message.providerPayload,
|
|
251
261
|
timestamp: message.timestamp,
|
|
252
262
|
};
|
|
@@ -1,8 +1,4 @@
|
|
|
1
|
-
|
|
2
|
-
SCOPE: The conversation's final {{#when retainedTail.count "==" 1}}{{retainedTail.role}} message stays{{else}}{{retainedTail.count}} messages, starting with a {{retainedTail.role}} message, stay{{/when}} in context verbatim after your summary. Summarize ONLY the history before those messages. You MUST NOT restate anything from those final messages — the reader sees them right after the summary — and you MUST treat them as the most recent state when describing progress and next steps.
|
|
3
|
-
{{else}}
|
|
4
|
-
SCOPE: The conversation above is the transcript to summarize. The API replaces everything before your summary with it, so nothing you leave out survives into the next context window.
|
|
5
|
-
{{/if}}
|
|
1
|
+
The conversation above is the complete history to summarize. The API replaces every message in this request with your summary; no message you leave out survives.
|
|
6
2
|
|
|
7
3
|
{{#if extraContext}}
|
|
8
4
|
{{extraContext}}
|
|
@@ -14,4 +10,4 @@ SCOPE: The conversation above is the transcript to summarize. The API replaces e
|
|
|
14
10
|
Additional focus: {{customInstructions}}
|
|
15
11
|
{{/if}}
|
|
16
12
|
|
|
17
|
-
You MUST NOT call any tools while writing the summary; respond with the summary text only.
|
|
13
|
+
You MUST NOT call any tools while writing the summary; respond with the summary text only.
|
|
@@ -18,7 +18,7 @@
|
|
|
18
18
|
* - `hasContextTokenUsage(usage)`: the report must carry usable context numbers.
|
|
19
19
|
*/
|
|
20
20
|
|
|
21
|
-
import type { AssistantMessage } from "@oh-my-pi/pi-ai";
|
|
21
|
+
import type { AssistantMessage, Message } from "@oh-my-pi/pi-ai";
|
|
22
22
|
import type { MessageCountOptions, Tokenizer } from "../tokenizer";
|
|
23
23
|
import type { AgentMessage } from "../types";
|
|
24
24
|
import { calculateContextTokens, hasContextTokenUsage } from "./compaction";
|
|
@@ -67,6 +67,36 @@ export function findTranscriptUsageAnchor(
|
|
|
67
67
|
return undefined;
|
|
68
68
|
}
|
|
69
69
|
|
|
70
|
+
/**
|
|
71
|
+
* Newest assistant turn in a provider request's `messages` whose usage still
|
|
72
|
+
* describes the prefix it sits on, or `undefined` when none does.
|
|
73
|
+
*
|
|
74
|
+
* Request contexts carry no compaction index, so staleness is read from the
|
|
75
|
+
* rewrite markers themselves: a compaction/branch summary or pruned tool result
|
|
76
|
+
* (`prunedAt`) replaced text that every report made at or before the rewrite
|
|
77
|
+
* already counted. A summary's rewrite time is its `timestamp` (commit time);
|
|
78
|
+
* its `historyRewriteAt` may be predated before a natively replayed retained
|
|
79
|
+
* tail so that tail's bound thinking survives, but the tail's usage still
|
|
80
|
+
* counted the summarized prefix.
|
|
81
|
+
*/
|
|
82
|
+
export function findRequestUsageAnchor(messages: readonly Message[]): TranscriptUsageAnchor | undefined {
|
|
83
|
+
let rewriteAt = Number.NEGATIVE_INFINITY;
|
|
84
|
+
let anchorIndex = -1;
|
|
85
|
+
let anchor: AssistantMessage | undefined;
|
|
86
|
+
for (let index = 0; index < messages.length; index++) {
|
|
87
|
+
const message = messages[index];
|
|
88
|
+
if (message.role === "user" && message.historyRewriteAt !== undefined) {
|
|
89
|
+
rewriteAt = Math.max(rewriteAt, message.historyRewriteAt, message.timestamp);
|
|
90
|
+
} else if (message.role === "toolResult" && message.prunedAt !== undefined) {
|
|
91
|
+
rewriteAt = Math.max(rewriteAt, message.prunedAt);
|
|
92
|
+
} else if (isTranscriptUsageAnchor(message) && message.timestamp > rewriteAt) {
|
|
93
|
+
anchorIndex = index;
|
|
94
|
+
anchor = message;
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
return anchor && { index: anchorIndex, message: anchor, tokens: calculateContextTokens(anchor.usage) };
|
|
98
|
+
}
|
|
99
|
+
|
|
70
100
|
/** Options for {@link estimateTranscriptTokens}. */
|
|
71
101
|
export interface TranscriptTokenOptions {
|
|
72
102
|
/**
|
package/src/index.ts
CHANGED
|
@@ -6,6 +6,8 @@ export * from "./agent-loop";
|
|
|
6
6
|
export * from "./append-only-context";
|
|
7
7
|
// Compaction
|
|
8
8
|
export * from "./compaction";
|
|
9
|
+
// Output cap sized to the remaining context window
|
|
10
|
+
export * from "./output-budget";
|
|
9
11
|
// Process-global pause gate
|
|
10
12
|
export * from "./pause";
|
|
11
13
|
// Proxy utilities
|
|
@@ -14,12 +16,16 @@ export * from "./proxy";
|
|
|
14
16
|
export * from "./replay-policy";
|
|
15
17
|
// Run-level telemetry collector + aggregators
|
|
16
18
|
export * from "./run-collector";
|
|
19
|
+
// Tool definitions remembered for Anthropic inactive-tool re-declaration
|
|
20
|
+
export * from "./sent-tool-definitions";
|
|
17
21
|
// Speculative execution coordinator
|
|
18
22
|
export * from "./speculative-execution";
|
|
19
23
|
// Telemetry
|
|
20
24
|
export * from "./telemetry";
|
|
21
25
|
// Thinking selectors
|
|
22
26
|
export * from "./thinking";
|
|
27
|
+
// Tool-context augmentation
|
|
28
|
+
export * from "./tool-context";
|
|
23
29
|
// Tokenizer choice
|
|
24
30
|
export * from "./tokenizer";
|
|
25
31
|
// Types
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Agent side of provider live steering ({@link LiveSteering}).
|
|
3
|
+
*
|
|
4
|
+
* A provider that can put user input into the response it is streaming (OpenAI
|
|
5
|
+
* Responses `response.steer`) pulls queued steering through a
|
|
6
|
+
* {@link LiveSteeringChannel}. The loop records what the provider accepted right
|
|
7
|
+
* after that response, so the transcript matches what the model saw; anything
|
|
8
|
+
* it declined is injected at the next boundary like ordinary steering.
|
|
9
|
+
*/
|
|
10
|
+
import type { LiveSteerClaim, LiveSteering, UserMessage } from "@oh-my-pi/pi-ai";
|
|
11
|
+
import { logger } from "@oh-my-pi/pi-utils";
|
|
12
|
+
import type { AgentMessage } from "./types";
|
|
13
|
+
|
|
14
|
+
/** Steering-queue access for one provider call, supplied by the agent loop. */
|
|
15
|
+
export interface LiveSteeringQueue {
|
|
16
|
+
/** Resolves once steering is queued or `signal` aborts; never consumes. */
|
|
17
|
+
wait(signal: AbortSignal): Promise<void>;
|
|
18
|
+
/** Dequeues the next steering batch. */
|
|
19
|
+
take(signal: AbortSignal): Promise<AgentMessage[]>;
|
|
20
|
+
/**
|
|
21
|
+
* Provider view of `messages` appended to the in-flight call's context, or
|
|
22
|
+
* `undefined` when that view is not purely user messages.
|
|
23
|
+
*/
|
|
24
|
+
toProvider(messages: AgentMessage[], signal: AbortSignal): Promise<UserMessage[] | undefined>;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/** One provider call's {@link LiveSteering} source. */
|
|
28
|
+
export class LiveSteeringChannel implements LiveSteering {
|
|
29
|
+
/** Steering the provider delivered into the in-flight response, in queue order. */
|
|
30
|
+
readonly accepted: AgentMessage[] = [];
|
|
31
|
+
/** Steering taken from the queue but not delivered; always queued after {@link accepted}. */
|
|
32
|
+
readonly deferred: AgentMessage[] = [];
|
|
33
|
+
readonly #queue: LiveSteeringQueue;
|
|
34
|
+
|
|
35
|
+
constructor(queue: LiveSteeringQueue) {
|
|
36
|
+
this.#queue = queue;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
wait(signal: AbortSignal): Promise<void> {
|
|
40
|
+
// Once input is deferred, later input must follow it at the boundary;
|
|
41
|
+
// delivering it live would reorder the user's messages.
|
|
42
|
+
if (this.deferred.length === 0) return this.#queue.wait(signal);
|
|
43
|
+
if (signal.aborted) return Promise.resolve();
|
|
44
|
+
const { promise, resolve } = Promise.withResolvers<void>();
|
|
45
|
+
signal.addEventListener("abort", () => resolve(), { once: true });
|
|
46
|
+
return promise;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
async claim(signal: AbortSignal): Promise<LiveSteerClaim | undefined> {
|
|
50
|
+
if (this.deferred.length > 0 || signal.aborted) return undefined;
|
|
51
|
+
let messages: AgentMessage[];
|
|
52
|
+
try {
|
|
53
|
+
messages = await this.#queue.take(signal);
|
|
54
|
+
} catch (error) {
|
|
55
|
+
// The queue restores what it could not hand over.
|
|
56
|
+
logger.debug("Live steering dequeue failed", {
|
|
57
|
+
error: error instanceof Error ? error.message : String(error),
|
|
58
|
+
});
|
|
59
|
+
return undefined;
|
|
60
|
+
}
|
|
61
|
+
if (messages.length === 0) return undefined;
|
|
62
|
+
let providerMessages: UserMessage[] | undefined;
|
|
63
|
+
try {
|
|
64
|
+
providerMessages = await this.#queue.toProvider(messages, signal);
|
|
65
|
+
} catch (error) {
|
|
66
|
+
logger.debug("Live steering conversion failed", {
|
|
67
|
+
error: error instanceof Error ? error.message : String(error),
|
|
68
|
+
});
|
|
69
|
+
}
|
|
70
|
+
if (!providerMessages) {
|
|
71
|
+
this.deferred.push(...messages);
|
|
72
|
+
return undefined;
|
|
73
|
+
}
|
|
74
|
+
let settled = false;
|
|
75
|
+
return {
|
|
76
|
+
messages: providerMessages,
|
|
77
|
+
accept: () => {
|
|
78
|
+
if (settled) return;
|
|
79
|
+
settled = true;
|
|
80
|
+
this.accepted.push(...messages);
|
|
81
|
+
},
|
|
82
|
+
reject: () => {
|
|
83
|
+
if (settled) return;
|
|
84
|
+
settled = true;
|
|
85
|
+
this.deferred.push(...messages);
|
|
86
|
+
},
|
|
87
|
+
};
|
|
88
|
+
}
|
|
89
|
+
}
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
import type { Context, Model, Tool } from "@oh-my-pi/pi-ai";
|
|
2
|
+
import { stopsOutputAtContextWindow } from "@oh-my-pi/pi-catalog/compat/output-limits";
|
|
3
|
+
import { stringifyJson } from "@oh-my-pi/pi-utils";
|
|
4
|
+
import { findRequestUsageAnchor } from "./compaction/transcript-tokens";
|
|
5
|
+
import type { Tokenizer } from "./tokenizer";
|
|
6
|
+
|
|
7
|
+
/** Smallest output cap {@link fitOutputTokensToContextWindow} will request. */
|
|
8
|
+
export const MIN_FITTED_OUTPUT_TOKENS = 1024;
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* Local counts are padded by 1/this before sizing the output cap: the
|
|
12
|
+
* provider's tokenizer can disagree with ours by a few percent, and
|
|
13
|
+
* undercounting reproduces the overflow this guards against. This is a
|
|
14
|
+
* tokenizer-error margin on locally counted text only (provider-reported
|
|
15
|
+
* usage is exact), not a context reserve; compaction's own reserve
|
|
16
|
+
* (`resolveBudgetReserveTokens`) still decides when to compact.
|
|
17
|
+
*/
|
|
18
|
+
const PROMPT_ESTIMATE_MARGIN_DIVISOR = 10;
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* Output cap for a request, so prompt plus output stays inside the model's
|
|
22
|
+
* context window.
|
|
23
|
+
*
|
|
24
|
+
* Chat Completions-style providers (DeepSeek, OpenAI, vLLM, ...) reject a
|
|
25
|
+
* request whose prompt tokens plus `max_tokens` exceed the window. Every
|
|
26
|
+
* request asks for `model.maxTokens` of output by default, so without this a
|
|
27
|
+
* large model output cap (DeepSeek V4: ~384k of a ~1M window) makes every
|
|
28
|
+
* request fail once the prompt passes window minus output cap, long before
|
|
29
|
+
* compaction triggers, and side turns (`/btw`, recaps) have no overflow
|
|
30
|
+
* recovery at all.
|
|
31
|
+
*
|
|
32
|
+
* The prompt size is the provider's own report from the newest trustworthy
|
|
33
|
+
* assistant turn (see {@link findRequestUsageAnchor}) plus a local count of
|
|
34
|
+
* only the messages appended after it; the whole context is counted locally
|
|
35
|
+
* only when no turn can anchor (fresh or freshly rewritten context).
|
|
36
|
+
*
|
|
37
|
+
* Returns `maxTokens` unchanged when the requested cap already fits, the
|
|
38
|
+
* model declares no window, the host ends generation at the window itself
|
|
39
|
+
* instead of rejecting the request (`stops-output-at-context-window`, e.g.
|
|
40
|
+
* Claude 4.5+ on the Claude API), or nothing would be requested (including an
|
|
41
|
+
* OpenRouter-hosted model with no caller cap: the transport omits the catalog
|
|
42
|
+
* default there so each upstream self-caps, and a fitted value would turn into
|
|
43
|
+
* an explicit cap that filters upstreams). Otherwise returns
|
|
44
|
+
* the remaining room (never below {@link MIN_FITTED_OUTPUT_TOKENS}); a
|
|
45
|
+
* prompt that fills the whole window still overflows and is left to the
|
|
46
|
+
* caller's compaction. Near a full window the floor means a turn can stop on
|
|
47
|
+
* `length` instead of failing with a 400.
|
|
48
|
+
*
|
|
49
|
+
* Lives here, not next to the default in pi-ai's `mapOptionsForApi`, because
|
|
50
|
+
* pi-ai has no tokenizer; callers apply it in their `streamFn` (coding-agent
|
|
51
|
+
* does so in its shared settings-aware wrapper).
|
|
52
|
+
*
|
|
53
|
+
* Not fixed: Anthropic budget-thinking transports raise `max_tokens` back to
|
|
54
|
+
* at least the thinking budget plus a fallback buffer downstream
|
|
55
|
+
* (`ensureMaxTokensForThinking`), so a fitted cap below that is overridden
|
|
56
|
+
* and the request can still exceed the window as before.
|
|
57
|
+
*/
|
|
58
|
+
export function fitOutputTokensToContextWindow(
|
|
59
|
+
model: Model,
|
|
60
|
+
context: Context,
|
|
61
|
+
maxTokens: number | undefined,
|
|
62
|
+
tokenizer: Tokenizer,
|
|
63
|
+
): number | undefined {
|
|
64
|
+
if (maxTokens === undefined && omitsDefaultOutputCap(model.compat)) return undefined;
|
|
65
|
+
const requested = maxTokens ?? model.maxTokens;
|
|
66
|
+
const contextWindow = model.contextWindow;
|
|
67
|
+
if (!requested || !contextWindow || contextWindow <= 0) return maxTokens;
|
|
68
|
+
if (stopsOutputAtContextWindow(model)) return maxTokens;
|
|
69
|
+
|
|
70
|
+
const room = contextWindow - countPromptTokens(context, tokenizer);
|
|
71
|
+
if (room >= requested) return maxTokens;
|
|
72
|
+
return Math.max(MIN_FITTED_OUTPUT_TOKENS, room);
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/** OpenRouter hosts drop the catalog default cap unless the endpoint always needs one. */
|
|
76
|
+
function omitsDefaultOutputCap(compat: Model["compat"] | undefined): boolean {
|
|
77
|
+
return (
|
|
78
|
+
compat !== undefined && "isOpenRouterHost" in compat && compat.isOpenRouterHost && !compat.alwaysSendMaxTokens
|
|
79
|
+
);
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* Framing tokens (system prompt, tool definitions) memoized per array: these
|
|
84
|
+
* are stable identities for the life of a turn, so side turns and repeat
|
|
85
|
+
* requests do not re-stringify and re-tokenize them. Length is part of the key
|
|
86
|
+
* to catch in-place growth.
|
|
87
|
+
*/
|
|
88
|
+
const framingCounts = new WeakMap<readonly unknown[], { tokenizer: Tokenizer; length: number; tokens: number }>();
|
|
89
|
+
|
|
90
|
+
function countFraming(items: readonly unknown[] | undefined, tokenizer: Tokenizer, fragments: () => string[]): number {
|
|
91
|
+
if (!items || items.length === 0) return 0;
|
|
92
|
+
const cached = framingCounts.get(items);
|
|
93
|
+
if (cached && cached.tokenizer === tokenizer && cached.length === items.length) return cached.tokens;
|
|
94
|
+
const tokens = tokenizer.countTokens(fragments());
|
|
95
|
+
framingCounts.set(items, { tokenizer, length: items.length, tokens });
|
|
96
|
+
return tokens;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
function toolFragments(tools: readonly Tool[]): string[] {
|
|
100
|
+
const fragments: string[] = [];
|
|
101
|
+
for (const tool of tools) fragments.push(tool.name, tool.description, stringifyJson(tool.parameters) ?? "");
|
|
102
|
+
return fragments;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
function withMargin(localTokens: number): number {
|
|
106
|
+
return localTokens + Math.ceil(localTokens / PROMPT_ESTIMATE_MARGIN_DIVISOR);
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
/** Provider-anchored prompt size; falls back to a full local count when nothing anchors. */
|
|
110
|
+
function countPromptTokens(context: Context, tokenizer: Tokenizer): number {
|
|
111
|
+
const { messages } = context;
|
|
112
|
+
const anchor = findRequestUsageAnchor(messages);
|
|
113
|
+
if (!anchor) return withMargin(countContextTokens(context, tokenizer));
|
|
114
|
+
let tail = 0;
|
|
115
|
+
for (let index = anchor.index + 1; index < messages.length; index++) {
|
|
116
|
+
tail += tokenizer.countMessage(messages[index]);
|
|
117
|
+
}
|
|
118
|
+
return anchor.tokens + withMargin(tail);
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
function countContextTokens(context: Context, tokenizer: Tokenizer): number {
|
|
122
|
+
const { systemPrompt, tools, inactiveTools } = context;
|
|
123
|
+
return (
|
|
124
|
+
countFraming(systemPrompt, tokenizer, () => [...(systemPrompt ?? [])]) +
|
|
125
|
+
countFraming(tools, tokenizer, () => toolFragments(tools ?? [])) +
|
|
126
|
+
// Anthropic replays retired tool definitions, so they are prompt too.
|
|
127
|
+
countFraming(inactiveTools, tokenizer, () => toolFragments(inactiveTools ?? [])) +
|
|
128
|
+
tokenizer.countMessages(context.messages)
|
|
129
|
+
);
|
|
130
|
+
}
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
import type { Message, Tool } from "@oh-my-pi/pi-ai";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Last wire definition this Agent sent for each tool name, so a provider that keeps
|
|
5
|
+
* withdrawn tools declared (Anthropic `tool_removal`) can re-declare them byte-identically.
|
|
6
|
+
* Used by prepareProviderCall and Agent.buildSideRequestContext.
|
|
7
|
+
*/
|
|
8
|
+
export class SentToolDefinitions {
|
|
9
|
+
#byName = new Map<string, Tool>();
|
|
10
|
+
|
|
11
|
+
/** Remember the definitions a request is about to send. */
|
|
12
|
+
record(tools: readonly Tool[]): void {
|
|
13
|
+
for (const tool of tools) this.#byName.set(tool.name, tool);
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* Definitions for names the latest `requestControls.tools.declared` in `messages` holds
|
|
18
|
+
* that are not in `active`; undefined when none. Names never sent by this Agent are
|
|
19
|
+
* skipped: the provider drops them from the declaration.
|
|
20
|
+
*/
|
|
21
|
+
inactiveFor(messages: readonly Message[], active: readonly Tool[]): Tool[] | undefined {
|
|
22
|
+
let declared: readonly string[] | undefined;
|
|
23
|
+
for (let index = messages.length - 1; index >= 0; index--) {
|
|
24
|
+
const message = messages[index];
|
|
25
|
+
if (message?.role === "assistant" && message.requestControls?.tools) {
|
|
26
|
+
declared = message.requestControls.tools.declared;
|
|
27
|
+
break;
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
if (!declared) return undefined;
|
|
31
|
+
const activeNames = new Set(active.map(tool => tool.name));
|
|
32
|
+
const inactive: Tool[] = [];
|
|
33
|
+
for (const name of declared) {
|
|
34
|
+
if (activeNames.has(name)) continue;
|
|
35
|
+
const tool = this.#byName.get(name);
|
|
36
|
+
if (tool) inactive.push(tool);
|
|
37
|
+
}
|
|
38
|
+
return inactive.length > 0 ? inactive : undefined;
|
|
39
|
+
}
|
|
40
|
+
}
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
import type { ToolResultMessage } from "@oh-my-pi/pi-ai";
|
|
2
|
+
import type { AgentMessage } from "./types";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Symbol-keyed carrier for passive context reported by a tool executed outside
|
|
6
|
+
* the agent loop (Cursor exec-channel dispatch). The executor attaches the
|
|
7
|
+
* joined context to the {@link ToolResultMessage} it returns; `Agent` reads it
|
|
8
|
+
* when the provider hands the result back and injects it after the buffered
|
|
9
|
+
* results. Symbol keys never serialize, so the context cannot leak into the
|
|
10
|
+
* persisted tool result.
|
|
11
|
+
*/
|
|
12
|
+
export const TOOL_RESULT_ADDITIONAL_CONTEXT = Symbol("tool-result-additional-context");
|
|
13
|
+
|
|
14
|
+
/** A tool result optionally carrying {@link TOOL_RESULT_ADDITIONAL_CONTEXT}. */
|
|
15
|
+
export type ToolResultWithAdditionalContext = ToolResultMessage & { [TOOL_RESULT_ADDITIONAL_CONTEXT]?: string };
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* True for a passive-context value worth delivering: a string with at least
|
|
19
|
+
* one non-whitespace character. Shared by every producer and aggregation site
|
|
20
|
+
* so blank values never produce a developer message.
|
|
21
|
+
*/
|
|
22
|
+
export function isNonBlankContext(value: unknown): value is string {
|
|
23
|
+
return typeof value === "string" && value.trim().length > 0;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* Join passive context values in order, dropping blanks. Returns undefined
|
|
28
|
+
* when nothing remains.
|
|
29
|
+
*/
|
|
30
|
+
export function joinAdditionalContext(values: Iterable<string | undefined>): string | undefined {
|
|
31
|
+
const kept: string[] = [];
|
|
32
|
+
for (const value of values) {
|
|
33
|
+
if (isNonBlankContext(value)) kept.push(value);
|
|
34
|
+
}
|
|
35
|
+
return kept.length > 0 ? kept.join("\n\n") : undefined;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Build the developer message that carries passive tool context to the next
|
|
40
|
+
* provider request. Emitted after the tool results it belongs to.
|
|
41
|
+
*/
|
|
42
|
+
export function createAdditionalContextMessage(text: string): AgentMessage {
|
|
43
|
+
return {
|
|
44
|
+
role: "developer",
|
|
45
|
+
content: [{ type: "text", text }],
|
|
46
|
+
attribution: "agent",
|
|
47
|
+
timestamp: Date.now(),
|
|
48
|
+
};
|
|
49
|
+
}
|