@oh-my-pi/pi-agent-core 18.2.11 → 18.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -43,9 +43,8 @@ import { ThinkingLevel } from "../thinking";
43
43
  import { Tokenizer } from "../tokenizer";
44
44
  import type { AgentMessage } from "../types";
45
45
  import {
46
- ANTHROPIC_COMPACTION_MIN_CONTEXT_TOKENS,
47
46
  buildAnthropicCompactionInstructions,
48
- describeRetainedTail,
47
+ findAnthropicCompactionCut,
49
48
  getPreservedAnthropicCompactionData,
50
49
  requestAnthropicNativeCompaction,
51
50
  shouldUseAnthropicNativeCompaction,
@@ -1252,6 +1251,8 @@ export interface CompactionPreparation {
1252
1251
  turnPrefixMessages: AgentMessage[];
1253
1252
  /** Messages kept in full after compaction (recent history) */
1254
1253
  recentMessages: AgentMessage[];
1254
+ /** Entry IDs parallel to recentMessages, for an Anthropic-safe keep-tail boundary. */
1255
+ recentEntryIds?: string[];
1255
1256
  /** Whether this is a split turn (cut point in middle of turn) */
1256
1257
  isSplitTurn: boolean;
1257
1258
  tokensBefore: number;
@@ -1380,16 +1381,22 @@ export function prepareCompaction(
1380
1381
  }
1381
1382
  }
1382
1383
  } else if (previousCompaction) {
1383
- // Local summaries exclude the retained tail, whose original entries precede
1384
- // the compaction record. Native replay already carries that tail. Only look
1385
- // backwards: advisor snapshots put all retained messages after the summary
1386
- // and may carry a keep ID from their previous, differently indexed snapshot.
1384
+ // Local and Anthropic summaries exclude the retained tail, whose
1385
+ // original entries precede the compaction record.
1387
1386
  for (let i = resetBoundaryIndex + 1; i < prevCompactionIndex; i++) {
1388
1387
  if (pathEntries[i].id === previousCompaction.firstKeptEntryId) {
1389
1388
  boundaryStart = i;
1390
1389
  break;
1391
1390
  }
1392
1391
  }
1392
+ if (previousCompaction.firstKeptEntryId === "" && previousCompaction.providerReplayThroughEntryId) {
1393
+ // An empty snapshot tail can still have turns appended during
1394
+ // background compaction; those start after the summarized snapshot.
1395
+ const snapshotIdx = pathEntries.findIndex(
1396
+ entry => entry.id === previousCompaction.providerReplayThroughEntryId,
1397
+ );
1398
+ if (snapshotIdx >= 0 && snapshotIdx < prevCompactionIndex) boundaryStart = snapshotIdx + 1;
1399
+ }
1393
1400
  }
1394
1401
 
1395
1402
  // Keep original IDs beside the converted messages so estimation, cutting,
@@ -1437,6 +1444,11 @@ export function prepareCompaction(
1437
1444
  return undefined;
1438
1445
  }
1439
1446
 
1447
+ const recentEntryIds: string[] = [];
1448
+ for (let i = cutPoint.firstKeptEntryIndex; i < compactionEntries.length; i++) {
1449
+ recentEntryIds.push(compactionEntries[i].id);
1450
+ }
1451
+
1440
1452
  // Extract file operations from messages and previous compaction
1441
1453
  const fileOps = extractFileOperations(messagesToSummarize, pathEntries, prevCompactionIndex);
1442
1454
 
@@ -1452,6 +1464,7 @@ export function prepareCompaction(
1452
1464
  messagesToSummarize,
1453
1465
  turnPrefixMessages,
1454
1466
  recentMessages,
1467
+ recentEntryIds,
1455
1468
  isSplitTurn: cutPoint.isSplitTurn,
1456
1469
  tokensBefore,
1457
1470
  previousSummary: previousCompaction?.summary,
@@ -1579,6 +1592,7 @@ export async function compact(
1579
1592
  messagesToSummarize,
1580
1593
  turnPrefixMessages,
1581
1594
  recentMessages,
1595
+ recentEntryIds,
1582
1596
  isSplitTurn,
1583
1597
  tokensBefore,
1584
1598
  previousSummary,
@@ -1826,35 +1840,19 @@ export async function compact(
1826
1840
  }
1827
1841
  }
1828
1842
 
1829
- // Anthropic server-side compaction: the live turn's request shape plus the
1830
- // compact edit, so the API summarizes from its cached prefix. The summary is
1831
- // real text, persisted both as the entry summary and as the native replay
1832
- // payload. A context below the API's trigger floor cannot compact remotely
1833
- // and takes the local summarizer instead — an eligibility boundary, not a
1834
- // failure.
1843
+ // On-demand compaction summarizes only the prefix; the tail is never sent
1844
+ // to this request and is replayed after the returned signed block.
1835
1845
  let nativeSummary: string | undefined;
1836
- let nativeEncryptedContent: string | undefined;
1846
+ let nativeSignature: string | undefined;
1847
+ let nativeFirstKeptEntryId = firstKeptEntryId;
1837
1848
  let nativeUsedTokens: number | undefined;
1838
- if (
1839
- !usedRemoteCompaction &&
1840
- settings.remoteEnabled !== false &&
1841
- shouldUseAnthropicNativeCompaction(model) &&
1842
- tokensBefore >= ANTHROPIC_COMPACTION_MIN_CONTEXT_TOKENS
1843
- ) {
1849
+ if (!usedRemoteCompaction && settings.remoteEnabled !== false && shouldUseAnthropicNativeCompaction(model)) {
1844
1850
  const previousNative = getPreservedAnthropicCompactionData(previousPreserveData);
1845
- // Lead with the previous summary exactly as the live context renders it:
1846
- // natively when this provider wrote it, as text otherwise. The request
1847
- // then shares the live turn's prefix byte-for-byte. A prior snapcompact
1848
- // archive is already merged into that summary text, so the archive
1849
- // migration message the OpenAI lanes carry is omitted here.
1850
- // The rewrite marker must precede every message this request replays —
1851
- // summarized history and retained tail alike — exactly like the live
1852
- // context rebuild predates its tail. A previous compaction's commit
1853
- // timestamp is newer than re-retained or re-summarized turns, so
1854
- // reusing it would strip their bound thinking only in this request,
1855
- // diverging from the cached live prefix (and possibly dropping below
1856
- // the trigger). Manually built preparations with no input at all fall
1857
- // back to the current time.
1851
+ // Lead with the previous summary as the live context renders it:
1852
+ // natively when this provider wrote it, as text otherwise.
1853
+ // A prior snapcompact archive is already merged into that summary
1854
+ // text. Predate the rewrite marker before all replayed messages so
1855
+ // retained thinking stays bound to the original prefix.
1858
1856
  const firstReplayed = messagesToSummarize[0] ?? turnPrefixMessages[0] ?? recentMessages[0];
1859
1857
  const previousSummaryAt =
1860
1858
  firstReplayed !== undefined ? new Date(firstReplayed.timestamp - 1).toISOString() : new Date().toISOString();
@@ -1866,6 +1864,7 @@ export async function compact(
1866
1864
  type: "anthropicCompaction",
1867
1865
  provider: previousNative.provider,
1868
1866
  content: previousNative.content,
1867
+ ...(previousNative.signature ? { signature: previousNative.signature } : {}),
1869
1868
  ...(previousNative.encryptedContent
1870
1869
  ? { encryptedContent: previousNative.encryptedContent }
1871
1870
  : {}),
@@ -1874,29 +1873,40 @@ export async function compact(
1874
1873
  : undefined,
1875
1874
  })
1876
1875
  : undefined;
1877
- const convertToLlm = summaryOptions.convertToLlm ?? defaultConvertToLlm;
1878
- const retainedTail = convertToLlm(recentMessages);
1879
- const messages = [
1880
- ...convertToLlm([
1881
- ...(previousSummaryMessage ? [previousSummaryMessage] : []),
1882
- ...messagesToSummarize,
1883
- ...turnPrefixMessages,
1884
- ]),
1885
- ...retainedTail,
1886
- ];
1876
+ const allMessages = [...messagesToSummarize, ...turnPrefixMessages, ...recentMessages];
1877
+ const originalCut = messagesToSummarize.length + turnPrefixMessages.length;
1878
+ const nativeCut = findAnthropicCompactionCut(allMessages, originalCut);
1879
+ // Hand-built preparations lacking entry IDs cannot move their persisted
1880
+ // boundary into the tail. Summarizing all is still safe.
1881
+ const safeCut =
1882
+ nativeCut > originalCut && nativeCut < allMessages.length && !recentEntryIds?.[nativeCut - originalCut]
1883
+ ? allMessages.length
1884
+ : nativeCut;
1885
+ nativeFirstKeptEntryId =
1886
+ safeCut === allMessages.length
1887
+ ? ""
1888
+ : safeCut === originalCut
1889
+ ? firstKeptEntryId
1890
+ : (recentEntryIds?.[safeCut - originalCut] ?? "");
1891
+ for (let i = originalCut; i < safeCut; i++) extractFileOpsFromMessage(allMessages[i], fileOps);
1892
+ const messages = (summaryOptions.convertToLlm ?? defaultConvertToLlm)([
1893
+ ...(previousSummaryMessage ? [previousSummaryMessage] : []),
1894
+ ...allMessages.slice(0, safeCut),
1895
+ ]);
1887
1896
  try {
1888
1897
  const remote = await requestAnthropicNativeCompaction(
1889
1898
  model,
1890
1899
  apiKey,
1891
1900
  {
1892
- systemPrompt: summaryOptions.remoteSystemPrompt ?? [SUMMARIZATION_SYSTEM_PROMPT],
1901
+ // The live prompt, not the local summarizer's synthetic system
1902
+ // prompt: kept thinking remains valid only under identical controls.
1903
+ systemPrompt: summaryOptions.remoteSystemPrompt ?? [],
1893
1904
  messages,
1894
1905
  tools: summaryOptions.tools,
1895
1906
  instructions: buildAnthropicCompactionInstructions(
1896
1907
  summaryOptions.promptOverride ?? SUMMARIZATION_PROMPT,
1897
1908
  customInstructions,
1898
1909
  formatAdditionalContext(summaryOptions.extraContext).trim() || undefined,
1899
- describeRetainedTail(retainedTail),
1900
1910
  ),
1901
1911
  maxTokens: Math.min(Math.floor(0.8 * reserveTokens), MAX_SUMMARY_TOKENS),
1902
1912
  reasoning: resolveCompactionEffort(model, summaryOptions.thinkingLevel),
@@ -1915,7 +1925,7 @@ export async function compact(
1915
1925
  },
1916
1926
  );
1917
1927
  nativeSummary = remote.content;
1918
- nativeEncryptedContent = remote.encryptedContent;
1928
+ nativeSignature = remote.signature;
1919
1929
  nativeUsedTokens = calculatePromptTokens(remote.usage);
1920
1930
  usedRemoteCompaction = true;
1921
1931
  } catch (err) {
@@ -2004,7 +2014,7 @@ export async function compact(
2004
2014
  summary = upsertFileOperations(summary, readFiles, modifiedFiles, fileOps.read);
2005
2015
  if (nativeSummary !== undefined) {
2006
2016
  // The replayed block stays byte-identical to the API's summary so it
2007
- // matches `encryptedContent`. The harness file lists above travel
2017
+ // matches its signature. The harness file lists above travel
2008
2018
  // separately: the converter replaces the summary message with the
2009
2019
  // block and skips its text, so they would otherwise be invisible to
2010
2020
  // this provider. Every other provider keeps reading the entry text.
@@ -2012,7 +2022,7 @@ export async function compact(
2012
2022
  preserveData = withAnthropicCompactionPreserveData(preserveData, {
2013
2023
  provider: model.provider,
2014
2024
  content: nativeSummary,
2015
- ...(nativeEncryptedContent ? { encryptedContent: nativeEncryptedContent } : {}),
2025
+ ...(nativeSignature ? { signature: nativeSignature } : {}),
2016
2026
  ...(filesText ? { filesText } : {}),
2017
2027
  model: model.id,
2018
2028
  usedTokens: nativeUsedTokens,
@@ -2034,7 +2044,7 @@ export async function compact(
2034
2044
  return {
2035
2045
  summary,
2036
2046
  shortSummary,
2037
- firstKeptEntryId,
2047
+ firstKeptEntryId: nativeSummary !== undefined ? nativeFirstKeptEntryId : firstKeptEntryId,
2038
2048
  tokensBefore,
2039
2049
  details: { readFiles, modifiedFiles } as CompactionDetails,
2040
2050
  preserveData: finalPreserveData,
@@ -64,6 +64,13 @@ export interface CompactionSummaryMessage {
64
64
  images?: ImageContent[];
65
65
  /** Post-pass dead-end warning attached to this compaction (progress guard). */
66
66
  warning?: string;
67
+ /**
68
+ * Thinking-binding rewrite marker when it must differ from `timestamp`: a
69
+ * natively replayed summary predates it before the retained tail so that
70
+ * tail's bound thinking stays valid. `timestamp` remains the commit time,
71
+ * which is what invalidates the tail's pre-compaction usage reports.
72
+ */
73
+ historyRewriteAt?: number;
67
74
  timestamp: number;
68
75
  }
69
76
 
@@ -135,6 +142,8 @@ export interface CompactionSummaryMessageOptions {
135
142
  method?: string;
136
143
  /** Estimated context tokens after the rewrite, for display alongside `tokensBefore`. */
137
144
  tokensAfter?: number;
145
+ /** See {@link CompactionSummaryMessage.historyRewriteAt}. */
146
+ historyRewriteAt?: number;
138
147
  }
139
148
 
140
149
  export function createCompactionSummaryMessage(
@@ -143,7 +152,7 @@ export function createCompactionSummaryMessage(
143
152
  timestamp: string,
144
153
  options: CompactionSummaryMessageOptions = {},
145
154
  ): CompactionSummaryMessage {
146
- const { shortSummary, providerPayload, images, blocks, warning, method, tokensAfter } = options;
155
+ const { shortSummary, providerPayload, images, blocks, warning, method, tokensAfter, historyRewriteAt } = options;
147
156
  const imageBlocks =
148
157
  blocks?.filter((block): block is ImageContent => block.type === "image") ??
149
158
  (images && images.length > 0 ? images : undefined);
@@ -158,6 +167,7 @@ export function createCompactionSummaryMessage(
158
167
  blocks: blocks && blocks.length > 0 ? blocks : undefined,
159
168
  images: imageBlocks && imageBlocks.length > 0 ? imageBlocks : undefined,
160
169
  warning,
170
+ historyRewriteAt,
161
171
  timestamp: new Date(timestamp).getTime(),
162
172
  };
163
173
  }
@@ -246,7 +256,7 @@ export function convertMessageToLlm(message: AgentMessage): Message | undefined
246
256
  ...(message.images ?? []),
247
257
  ],
248
258
  attribution: "agent",
249
- historyRewriteAt: message.timestamp,
259
+ historyRewriteAt: message.historyRewriteAt ?? message.timestamp,
250
260
  providerPayload: message.providerPayload,
251
261
  timestamp: message.timestamp,
252
262
  };
@@ -1,8 +1,4 @@
1
- {{#if retainedTail}}
2
- SCOPE: The conversation's final {{#when retainedTail.count "==" 1}}{{retainedTail.role}} message stays{{else}}{{retainedTail.count}} messages, starting with a {{retainedTail.role}} message, stay{{/when}} in context verbatim after your summary. Summarize ONLY the history before those messages. You MUST NOT restate anything from those final messages — the reader sees them right after the summary — and you MUST treat them as the most recent state when describing progress and next steps.
3
- {{else}}
4
- SCOPE: The conversation above is the transcript to summarize. The API replaces everything before your summary with it, so nothing you leave out survives into the next context window.
5
- {{/if}}
1
+ The conversation above is the complete history to summarize. The API replaces every message in this request with your summary; no message you leave out survives.
6
2
 
7
3
  {{#if extraContext}}
8
4
  {{extraContext}}
@@ -14,4 +10,4 @@ SCOPE: The conversation above is the transcript to summarize. The API replaces e
14
10
  Additional focus: {{customInstructions}}
15
11
  {{/if}}
16
12
 
17
- You MUST NOT call any tools while writing the summary; respond with the summary text only.
13
+ You MUST NOT call any tools while writing the summary; respond with the summary text only.
@@ -18,7 +18,7 @@
18
18
  * - `hasContextTokenUsage(usage)`: the report must carry usable context numbers.
19
19
  */
20
20
 
21
- import type { AssistantMessage } from "@oh-my-pi/pi-ai";
21
+ import type { AssistantMessage, Message } from "@oh-my-pi/pi-ai";
22
22
  import type { MessageCountOptions, Tokenizer } from "../tokenizer";
23
23
  import type { AgentMessage } from "../types";
24
24
  import { calculateContextTokens, hasContextTokenUsage } from "./compaction";
@@ -67,6 +67,36 @@ export function findTranscriptUsageAnchor(
67
67
  return undefined;
68
68
  }
69
69
 
70
+ /**
71
+ * Newest assistant turn in a provider request's `messages` whose usage still
72
+ * describes the prefix it sits on, or `undefined` when none does.
73
+ *
74
+ * Request contexts carry no compaction index, so staleness is read from the
75
+ * rewrite markers themselves: a compaction/branch summary or pruned tool result
76
+ * (`prunedAt`) replaced text that every report made at or before the rewrite
77
+ * already counted. A summary's rewrite time is its `timestamp` (commit time);
78
+ * its `historyRewriteAt` may be predated before a natively replayed retained
79
+ * tail so that tail's bound thinking survives, but the tail's usage still
80
+ * counted the summarized prefix.
81
+ */
82
+ export function findRequestUsageAnchor(messages: readonly Message[]): TranscriptUsageAnchor | undefined {
83
+ let rewriteAt = Number.NEGATIVE_INFINITY;
84
+ let anchorIndex = -1;
85
+ let anchor: AssistantMessage | undefined;
86
+ for (let index = 0; index < messages.length; index++) {
87
+ const message = messages[index];
88
+ if (message.role === "user" && message.historyRewriteAt !== undefined) {
89
+ rewriteAt = Math.max(rewriteAt, message.historyRewriteAt, message.timestamp);
90
+ } else if (message.role === "toolResult" && message.prunedAt !== undefined) {
91
+ rewriteAt = Math.max(rewriteAt, message.prunedAt);
92
+ } else if (isTranscriptUsageAnchor(message) && message.timestamp > rewriteAt) {
93
+ anchorIndex = index;
94
+ anchor = message;
95
+ }
96
+ }
97
+ return anchor && { index: anchorIndex, message: anchor, tokens: calculateContextTokens(anchor.usage) };
98
+ }
99
+
70
100
  /** Options for {@link estimateTranscriptTokens}. */
71
101
  export interface TranscriptTokenOptions {
72
102
  /**
package/src/index.ts CHANGED
@@ -6,6 +6,8 @@ export * from "./agent-loop";
6
6
  export * from "./append-only-context";
7
7
  // Compaction
8
8
  export * from "./compaction";
9
+ // Output cap sized to the remaining context window
10
+ export * from "./output-budget";
9
11
  // Process-global pause gate
10
12
  export * from "./pause";
11
13
  // Proxy utilities
@@ -14,12 +16,16 @@ export * from "./proxy";
14
16
  export * from "./replay-policy";
15
17
  // Run-level telemetry collector + aggregators
16
18
  export * from "./run-collector";
19
+ // Tool definitions remembered for Anthropic inactive-tool re-declaration
20
+ export * from "./sent-tool-definitions";
17
21
  // Speculative execution coordinator
18
22
  export * from "./speculative-execution";
19
23
  // Telemetry
20
24
  export * from "./telemetry";
21
25
  // Thinking selectors
22
26
  export * from "./thinking";
27
+ // Tool-context augmentation
28
+ export * from "./tool-context";
23
29
  // Tokenizer choice
24
30
  export * from "./tokenizer";
25
31
  // Types
@@ -0,0 +1,89 @@
1
+ /**
2
+ * Agent side of provider live steering ({@link LiveSteering}).
3
+ *
4
+ * A provider that can put user input into the response it is streaming (OpenAI
5
+ * Responses `response.steer`) pulls queued steering through a
6
+ * {@link LiveSteeringChannel}. The loop records what the provider accepted right
7
+ * after that response, so the transcript matches what the model saw; anything
8
+ * it declined is injected at the next boundary like ordinary steering.
9
+ */
10
+ import type { LiveSteerClaim, LiveSteering, UserMessage } from "@oh-my-pi/pi-ai";
11
+ import { logger } from "@oh-my-pi/pi-utils";
12
+ import type { AgentMessage } from "./types";
13
+
14
+ /** Steering-queue access for one provider call, supplied by the agent loop. */
15
+ export interface LiveSteeringQueue {
16
+ /** Resolves once steering is queued or `signal` aborts; never consumes. */
17
+ wait(signal: AbortSignal): Promise<void>;
18
+ /** Dequeues the next steering batch. */
19
+ take(signal: AbortSignal): Promise<AgentMessage[]>;
20
+ /**
21
+ * Provider view of `messages` appended to the in-flight call's context, or
22
+ * `undefined` when that view is not purely user messages.
23
+ */
24
+ toProvider(messages: AgentMessage[], signal: AbortSignal): Promise<UserMessage[] | undefined>;
25
+ }
26
+
27
+ /** One provider call's {@link LiveSteering} source. */
28
+ export class LiveSteeringChannel implements LiveSteering {
29
+ /** Steering the provider delivered into the in-flight response, in queue order. */
30
+ readonly accepted: AgentMessage[] = [];
31
+ /** Steering taken from the queue but not delivered; always queued after {@link accepted}. */
32
+ readonly deferred: AgentMessage[] = [];
33
+ readonly #queue: LiveSteeringQueue;
34
+
35
+ constructor(queue: LiveSteeringQueue) {
36
+ this.#queue = queue;
37
+ }
38
+
39
+ wait(signal: AbortSignal): Promise<void> {
40
+ // Once input is deferred, later input must follow it at the boundary;
41
+ // delivering it live would reorder the user's messages.
42
+ if (this.deferred.length === 0) return this.#queue.wait(signal);
43
+ if (signal.aborted) return Promise.resolve();
44
+ const { promise, resolve } = Promise.withResolvers<void>();
45
+ signal.addEventListener("abort", () => resolve(), { once: true });
46
+ return promise;
47
+ }
48
+
49
+ async claim(signal: AbortSignal): Promise<LiveSteerClaim | undefined> {
50
+ if (this.deferred.length > 0 || signal.aborted) return undefined;
51
+ let messages: AgentMessage[];
52
+ try {
53
+ messages = await this.#queue.take(signal);
54
+ } catch (error) {
55
+ // The queue restores what it could not hand over.
56
+ logger.debug("Live steering dequeue failed", {
57
+ error: error instanceof Error ? error.message : String(error),
58
+ });
59
+ return undefined;
60
+ }
61
+ if (messages.length === 0) return undefined;
62
+ let providerMessages: UserMessage[] | undefined;
63
+ try {
64
+ providerMessages = await this.#queue.toProvider(messages, signal);
65
+ } catch (error) {
66
+ logger.debug("Live steering conversion failed", {
67
+ error: error instanceof Error ? error.message : String(error),
68
+ });
69
+ }
70
+ if (!providerMessages) {
71
+ this.deferred.push(...messages);
72
+ return undefined;
73
+ }
74
+ let settled = false;
75
+ return {
76
+ messages: providerMessages,
77
+ accept: () => {
78
+ if (settled) return;
79
+ settled = true;
80
+ this.accepted.push(...messages);
81
+ },
82
+ reject: () => {
83
+ if (settled) return;
84
+ settled = true;
85
+ this.deferred.push(...messages);
86
+ },
87
+ };
88
+ }
89
+ }
@@ -0,0 +1,130 @@
1
+ import type { Context, Model, Tool } from "@oh-my-pi/pi-ai";
2
+ import { stopsOutputAtContextWindow } from "@oh-my-pi/pi-catalog/compat/output-limits";
3
+ import { stringifyJson } from "@oh-my-pi/pi-utils";
4
+ import { findRequestUsageAnchor } from "./compaction/transcript-tokens";
5
+ import type { Tokenizer } from "./tokenizer";
6
+
7
+ /** Smallest output cap {@link fitOutputTokensToContextWindow} will request. */
8
+ export const MIN_FITTED_OUTPUT_TOKENS = 1024;
9
+
10
+ /**
11
+ * Local counts are padded by 1/this before sizing the output cap: the
12
+ * provider's tokenizer can disagree with ours by a few percent, and
13
+ * undercounting reproduces the overflow this guards against. This is a
14
+ * tokenizer-error margin on locally counted text only (provider-reported
15
+ * usage is exact), not a context reserve; compaction's own reserve
16
+ * (`resolveBudgetReserveTokens`) still decides when to compact.
17
+ */
18
+ const PROMPT_ESTIMATE_MARGIN_DIVISOR = 10;
19
+
20
+ /**
21
+ * Output cap for a request, so prompt plus output stays inside the model's
22
+ * context window.
23
+ *
24
+ * Chat Completions-style providers (DeepSeek, OpenAI, vLLM, ...) reject a
25
+ * request whose prompt tokens plus `max_tokens` exceed the window. Every
26
+ * request asks for `model.maxTokens` of output by default, so without this a
27
+ * large model output cap (DeepSeek V4: ~384k of a ~1M window) makes every
28
+ * request fail once the prompt passes window minus output cap, long before
29
+ * compaction triggers, and side turns (`/btw`, recaps) have no overflow
30
+ * recovery at all.
31
+ *
32
+ * The prompt size is the provider's own report from the newest trustworthy
33
+ * assistant turn (see {@link findRequestUsageAnchor}) plus a local count of
34
+ * only the messages appended after it; the whole context is counted locally
35
+ * only when no turn can anchor (fresh or freshly rewritten context).
36
+ *
37
+ * Returns `maxTokens` unchanged when the requested cap already fits, the
38
+ * model declares no window, the host ends generation at the window itself
39
+ * instead of rejecting the request (`stops-output-at-context-window`, e.g.
40
+ * Claude 4.5+ on the Claude API), or nothing would be requested (including an
41
+ * OpenRouter-hosted model with no caller cap: the transport omits the catalog
42
+ * default there so each upstream self-caps, and a fitted value would turn into
43
+ * an explicit cap that filters upstreams). Otherwise returns
44
+ * the remaining room (never below {@link MIN_FITTED_OUTPUT_TOKENS}); a
45
+ * prompt that fills the whole window still overflows and is left to the
46
+ * caller's compaction. Near a full window the floor means a turn can stop on
47
+ * `length` instead of failing with a 400.
48
+ *
49
+ * Lives here, not next to the default in pi-ai's `mapOptionsForApi`, because
50
+ * pi-ai has no tokenizer; callers apply it in their `streamFn` (coding-agent
51
+ * does so in its shared settings-aware wrapper).
52
+ *
53
+ * Not fixed: Anthropic budget-thinking transports raise `max_tokens` back to
54
+ * at least the thinking budget plus a fallback buffer downstream
55
+ * (`ensureMaxTokensForThinking`), so a fitted cap below that is overridden
56
+ * and the request can still exceed the window as before.
57
+ */
58
+ export function fitOutputTokensToContextWindow(
59
+ model: Model,
60
+ context: Context,
61
+ maxTokens: number | undefined,
62
+ tokenizer: Tokenizer,
63
+ ): number | undefined {
64
+ if (maxTokens === undefined && omitsDefaultOutputCap(model.compat)) return undefined;
65
+ const requested = maxTokens ?? model.maxTokens;
66
+ const contextWindow = model.contextWindow;
67
+ if (!requested || !contextWindow || contextWindow <= 0) return maxTokens;
68
+ if (stopsOutputAtContextWindow(model)) return maxTokens;
69
+
70
+ const room = contextWindow - countPromptTokens(context, tokenizer);
71
+ if (room >= requested) return maxTokens;
72
+ return Math.max(MIN_FITTED_OUTPUT_TOKENS, room);
73
+ }
74
+
75
+ /** OpenRouter hosts drop the catalog default cap unless the endpoint always needs one. */
76
+ function omitsDefaultOutputCap(compat: Model["compat"] | undefined): boolean {
77
+ return (
78
+ compat !== undefined && "isOpenRouterHost" in compat && compat.isOpenRouterHost && !compat.alwaysSendMaxTokens
79
+ );
80
+ }
81
+
82
+ /**
83
+ * Framing tokens (system prompt, tool definitions) memoized per array: these
84
+ * are stable identities for the life of a turn, so side turns and repeat
85
+ * requests do not re-stringify and re-tokenize them. Length is part of the key
86
+ * to catch in-place growth.
87
+ */
88
+ const framingCounts = new WeakMap<readonly unknown[], { tokenizer: Tokenizer; length: number; tokens: number }>();
89
+
90
+ function countFraming(items: readonly unknown[] | undefined, tokenizer: Tokenizer, fragments: () => string[]): number {
91
+ if (!items || items.length === 0) return 0;
92
+ const cached = framingCounts.get(items);
93
+ if (cached && cached.tokenizer === tokenizer && cached.length === items.length) return cached.tokens;
94
+ const tokens = tokenizer.countTokens(fragments());
95
+ framingCounts.set(items, { tokenizer, length: items.length, tokens });
96
+ return tokens;
97
+ }
98
+
99
+ function toolFragments(tools: readonly Tool[]): string[] {
100
+ const fragments: string[] = [];
101
+ for (const tool of tools) fragments.push(tool.name, tool.description, stringifyJson(tool.parameters) ?? "");
102
+ return fragments;
103
+ }
104
+
105
+ function withMargin(localTokens: number): number {
106
+ return localTokens + Math.ceil(localTokens / PROMPT_ESTIMATE_MARGIN_DIVISOR);
107
+ }
108
+
109
+ /** Provider-anchored prompt size; falls back to a full local count when nothing anchors. */
110
+ function countPromptTokens(context: Context, tokenizer: Tokenizer): number {
111
+ const { messages } = context;
112
+ const anchor = findRequestUsageAnchor(messages);
113
+ if (!anchor) return withMargin(countContextTokens(context, tokenizer));
114
+ let tail = 0;
115
+ for (let index = anchor.index + 1; index < messages.length; index++) {
116
+ tail += tokenizer.countMessage(messages[index]);
117
+ }
118
+ return anchor.tokens + withMargin(tail);
119
+ }
120
+
121
+ function countContextTokens(context: Context, tokenizer: Tokenizer): number {
122
+ const { systemPrompt, tools, inactiveTools } = context;
123
+ return (
124
+ countFraming(systemPrompt, tokenizer, () => [...(systemPrompt ?? [])]) +
125
+ countFraming(tools, tokenizer, () => toolFragments(tools ?? [])) +
126
+ // Anthropic replays retired tool definitions, so they are prompt too.
127
+ countFraming(inactiveTools, tokenizer, () => toolFragments(inactiveTools ?? [])) +
128
+ tokenizer.countMessages(context.messages)
129
+ );
130
+ }
@@ -0,0 +1,40 @@
1
+ import type { Message, Tool } from "@oh-my-pi/pi-ai";
2
+
3
+ /**
4
+ * Last wire definition this Agent sent for each tool name, so a provider that keeps
5
+ * withdrawn tools declared (Anthropic `tool_removal`) can re-declare them byte-identically.
6
+ * Used by prepareProviderCall and Agent.buildSideRequestContext.
7
+ */
8
+ export class SentToolDefinitions {
9
+ #byName = new Map<string, Tool>();
10
+
11
+ /** Remember the definitions a request is about to send. */
12
+ record(tools: readonly Tool[]): void {
13
+ for (const tool of tools) this.#byName.set(tool.name, tool);
14
+ }
15
+
16
+ /**
17
+ * Definitions for names the latest `requestControls.tools.declared` in `messages` holds
18
+ * that are not in `active`; undefined when none. Names never sent by this Agent are
19
+ * skipped: the provider drops them from the declaration.
20
+ */
21
+ inactiveFor(messages: readonly Message[], active: readonly Tool[]): Tool[] | undefined {
22
+ let declared: readonly string[] | undefined;
23
+ for (let index = messages.length - 1; index >= 0; index--) {
24
+ const message = messages[index];
25
+ if (message?.role === "assistant" && message.requestControls?.tools) {
26
+ declared = message.requestControls.tools.declared;
27
+ break;
28
+ }
29
+ }
30
+ if (!declared) return undefined;
31
+ const activeNames = new Set(active.map(tool => tool.name));
32
+ const inactive: Tool[] = [];
33
+ for (const name of declared) {
34
+ if (activeNames.has(name)) continue;
35
+ const tool = this.#byName.get(name);
36
+ if (tool) inactive.push(tool);
37
+ }
38
+ return inactive.length > 0 ? inactive : undefined;
39
+ }
40
+ }
@@ -0,0 +1,49 @@
1
+ import type { ToolResultMessage } from "@oh-my-pi/pi-ai";
2
+ import type { AgentMessage } from "./types";
3
+
4
+ /**
5
+ * Symbol-keyed carrier for passive context reported by a tool executed outside
6
+ * the agent loop (Cursor exec-channel dispatch). The executor attaches the
7
+ * joined context to the {@link ToolResultMessage} it returns; `Agent` reads it
8
+ * when the provider hands the result back and injects it after the buffered
9
+ * results. Symbol keys never serialize, so the context cannot leak into the
10
+ * persisted tool result.
11
+ */
12
+ export const TOOL_RESULT_ADDITIONAL_CONTEXT = Symbol("tool-result-additional-context");
13
+
14
+ /** A tool result optionally carrying {@link TOOL_RESULT_ADDITIONAL_CONTEXT}. */
15
+ export type ToolResultWithAdditionalContext = ToolResultMessage & { [TOOL_RESULT_ADDITIONAL_CONTEXT]?: string };
16
+
17
+ /**
18
+ * True for a passive-context value worth delivering: a string with at least
19
+ * one non-whitespace character. Shared by every producer and aggregation site
20
+ * so blank values never produce a developer message.
21
+ */
22
+ export function isNonBlankContext(value: unknown): value is string {
23
+ return typeof value === "string" && value.trim().length > 0;
24
+ }
25
+
26
+ /**
27
+ * Join passive context values in order, dropping blanks. Returns undefined
28
+ * when nothing remains.
29
+ */
30
+ export function joinAdditionalContext(values: Iterable<string | undefined>): string | undefined {
31
+ const kept: string[] = [];
32
+ for (const value of values) {
33
+ if (isNonBlankContext(value)) kept.push(value);
34
+ }
35
+ return kept.length > 0 ? kept.join("\n\n") : undefined;
36
+ }
37
+
38
+ /**
39
+ * Build the developer message that carries passive tool context to the next
40
+ * provider request. Emitted after the tool results it belongs to.
41
+ */
42
+ export function createAdditionalContextMessage(text: string): AgentMessage {
43
+ return {
44
+ role: "developer",
45
+ content: [{ type: "text", text }],
46
+ attribution: "agent",
47
+ timestamp: Date.now(),
48
+ };
49
+ }