@gajae-code/agent-core 0.11.1 → 0.11.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +7 -0
- package/dist/types/agent-loop.d.ts +44 -0
- package/dist/types/agent.d.ts +8 -0
- package/dist/types/compaction/compaction.d.ts +6 -2
- package/dist/types/compaction/pruning.d.ts +6 -2
- package/dist/types/proxy.d.ts +11 -0
- package/dist/types/types.d.ts +8 -2
- package/package.json +4 -4
- package/src/agent-loop.ts +504 -60
- package/src/agent.ts +36 -3
- package/src/compaction/compaction.ts +45 -89
- package/src/compaction/pruning.ts +144 -11
- package/src/proxy.ts +82 -2
- package/src/types.ts +4 -1
package/src/agent.ts
CHANGED
|
@@ -308,6 +308,11 @@ interface CursorToolResultEntry {
|
|
|
308
308
|
textLengthAtCall: number;
|
|
309
309
|
}
|
|
310
310
|
|
|
311
|
+
export type AgentQueueSnapshot = {
|
|
312
|
+
steering: AgentMessage[];
|
|
313
|
+
followUp: AgentMessage[];
|
|
314
|
+
};
|
|
315
|
+
|
|
311
316
|
export class Agent {
|
|
312
317
|
#state: AgentState = {
|
|
313
318
|
systemPrompt: [],
|
|
@@ -1009,6 +1014,20 @@ export class Agent {
|
|
|
1009
1014
|
this.#followUpQueue = [...messages, ...this.#followUpQueue];
|
|
1010
1015
|
}
|
|
1011
1016
|
|
|
1017
|
+
/** Snapshot both executable queues as one atomic session-level view. */
|
|
1018
|
+
snapshotQueues(): AgentQueueSnapshot {
|
|
1019
|
+
return {
|
|
1020
|
+
steering: this.#steeringQueue.slice(),
|
|
1021
|
+
followUp: this.#followUpQueue.slice(),
|
|
1022
|
+
};
|
|
1023
|
+
}
|
|
1024
|
+
|
|
1025
|
+
/** Replace both executable queues with a prior snapshot. */
|
|
1026
|
+
restoreQueues(snapshot: AgentQueueSnapshot): void {
|
|
1027
|
+
this.#steeringQueue = snapshot.steering.slice();
|
|
1028
|
+
this.#followUpQueue = snapshot.followUp.slice();
|
|
1029
|
+
}
|
|
1030
|
+
|
|
1012
1031
|
#dequeueSteeringMessages(): AgentMessage[] {
|
|
1013
1032
|
if (this.#steeringMode === "one-at-a-time") {
|
|
1014
1033
|
if (this.#steeringQueue.length > 0) {
|
|
@@ -1607,7 +1626,7 @@ export class Agent {
|
|
|
1607
1626
|
this.requestRunTerminal(managedLogicalRunOwner ?? runId, managedDecision.terminal);
|
|
1608
1627
|
} else if (managedOutcome.type === "run_terminal") {
|
|
1609
1628
|
this.requestRunTerminal(managedLogicalRunOwner ?? runId, { stopReason: managedOutcome.reason });
|
|
1610
|
-
} else if (managedDecision?.type !== "retry") {
|
|
1629
|
+
} else if (managedDecision?.type !== "retry" && managedDecision?.type !== "maintenance") {
|
|
1611
1630
|
this.#finalizeRun(managedLogicalRunOwner ?? runId);
|
|
1612
1631
|
}
|
|
1613
1632
|
}
|
|
@@ -1660,7 +1679,10 @@ export class Agent {
|
|
|
1660
1679
|
});
|
|
1661
1680
|
} finally {
|
|
1662
1681
|
let continuation: ManagedAttemptContinuation | undefined;
|
|
1663
|
-
if (
|
|
1682
|
+
if (
|
|
1683
|
+
managedOutcome?.type !== "run_terminal" &&
|
|
1684
|
+
(managedDecision?.type === "retry" || managedDecision?.type === "maintenance")
|
|
1685
|
+
) {
|
|
1664
1686
|
continuation = managedDecision.continuation;
|
|
1665
1687
|
}
|
|
1666
1688
|
const ownership: ManagedAttemptContinuationOwnership = {
|
|
@@ -1691,7 +1713,18 @@ export class Agent {
|
|
|
1691
1713
|
if (continuation && ownership.isCurrent()) {
|
|
1692
1714
|
try {
|
|
1693
1715
|
await continuation(ownership);
|
|
1694
|
-
if (
|
|
1716
|
+
if (
|
|
1717
|
+
managedDecision?.type === "maintenance" &&
|
|
1718
|
+
this.#terminalizedLogicalRunIds.has(managedLogicalRunOwner ?? runId) &&
|
|
1719
|
+
this.#managedLogicalRunOwner === managedLogicalRunOwner
|
|
1720
|
+
) {
|
|
1721
|
+
this.#managedLogicalRunOwner = undefined;
|
|
1722
|
+
}
|
|
1723
|
+
if (
|
|
1724
|
+
managedDecision?.type !== "maintenance" &&
|
|
1725
|
+
this.#activeRunId === undefined &&
|
|
1726
|
+
this.#managedLogicalRunOwner === managedLogicalRunOwner
|
|
1727
|
+
) {
|
|
1695
1728
|
this.#managedLogicalRunOwner = undefined;
|
|
1696
1729
|
}
|
|
1697
1730
|
} catch (err) {
|
|
@@ -29,7 +29,6 @@ import {
|
|
|
29
29
|
withOpenAiRemoteCompactionPreserveData,
|
|
30
30
|
} from "./openai";
|
|
31
31
|
import autoHandoffThresholdFocusPrompt from "./prompts/auto-handoff-threshold-focus.md" with { type: "text" };
|
|
32
|
-
import compactionShortSummaryPrompt from "./prompts/compaction-short-summary.md" with { type: "text" };
|
|
33
32
|
import compactionSummaryPrompt from "./prompts/compaction-summary.md" with { type: "text" };
|
|
34
33
|
import compactionTurnPrefixPrompt from "./prompts/compaction-turn-prefix.md" with { type: "text" };
|
|
35
34
|
import compactionUpdateSummaryPrompt from "./prompts/compaction-update-summary.md" with { type: "text" };
|
|
@@ -510,7 +509,8 @@ function collectMessageFragments(message: AgentMessage): { fragments: string[];
|
|
|
510
509
|
}
|
|
511
510
|
|
|
512
511
|
switch (message.role) {
|
|
513
|
-
case "user":
|
|
512
|
+
case "user":
|
|
513
|
+
case "custom": {
|
|
514
514
|
const content = (message as { content: string | Array<{ type: string; text?: string }> }).content;
|
|
515
515
|
if (typeof content === "string") {
|
|
516
516
|
fragments.push(content);
|
|
@@ -710,8 +710,6 @@ export function findCutPoint(
|
|
|
710
710
|
|
|
711
711
|
for (let i = endIndex - 1; i >= startIndex; i--) {
|
|
712
712
|
const entry = entries[i];
|
|
713
|
-
if (entry.type !== "message") continue;
|
|
714
|
-
|
|
715
713
|
// Estimate this message's size
|
|
716
714
|
const messageTokens = estimateEntryTokens(entry);
|
|
717
715
|
accumulatedTokens += messageTokens;
|
|
@@ -769,8 +767,6 @@ const SUMMARIZATION_PROMPT = prompt.render(compactionSummaryPrompt);
|
|
|
769
767
|
|
|
770
768
|
const UPDATE_SUMMARIZATION_PROMPT = prompt.render(compactionUpdateSummaryPrompt);
|
|
771
769
|
|
|
772
|
-
const SHORT_SUMMARY_PROMPT = prompt.render(compactionShortSummaryPrompt);
|
|
773
|
-
|
|
774
770
|
const HANDOFF_DOCUMENT_PROMPT = prompt.render(handoffDocumentPrompt);
|
|
775
771
|
|
|
776
772
|
export const AUTO_HANDOFF_THRESHOLD_FOCUS = prompt.render(autoHandoffThresholdFocusPrompt);
|
|
@@ -796,8 +792,7 @@ export interface SummaryOptions {
|
|
|
796
792
|
/**
|
|
797
793
|
* Optional telemetry handle. When provided, every LLM call emitted during
|
|
798
794
|
* compaction is wrapped in an OTEL chat span tagged with
|
|
799
|
-
* `pi.gen_ai.oneshot.kind` (`compaction_summary
|
|
800
|
-
* or `compaction_turn_prefix`). `undefined` keeps the call paths zero-cost.
|
|
795
|
+
* `pi.gen_ai.oneshot.kind` (`compaction_summary` or `compaction_turn_prefix`).
|
|
801
796
|
*/
|
|
802
797
|
telemetry?: AgentTelemetry;
|
|
803
798
|
authCredentialType?: "api_key" | "oauth";
|
|
@@ -1054,66 +1049,11 @@ export async function generateHandoff(
|
|
|
1054
1049
|
.join("\n");
|
|
1055
1050
|
}
|
|
1056
1051
|
|
|
1057
|
-
|
|
1058
|
-
|
|
1059
|
-
|
|
1060
|
-
|
|
1061
|
-
|
|
1062
|
-
apiKey: string,
|
|
1063
|
-
signal?: AbortSignal,
|
|
1064
|
-
options?: SummaryOptions,
|
|
1065
|
-
): Promise<string> {
|
|
1066
|
-
const maxTokens = Math.min(512, Math.floor(0.2 * reserveTokens));
|
|
1067
|
-
const llmMessages = (options?.convertToLlm ?? convertToLlm)(recentMessages);
|
|
1068
|
-
const conversationText = boundConversationTextForSummary(serializeConversation(llmMessages), model, maxTokens);
|
|
1069
|
-
|
|
1070
|
-
let promptText = `<conversation>\n${conversationText}\n</conversation>\n\n`;
|
|
1071
|
-
if (historySummary) {
|
|
1072
|
-
promptText += `<previous-summary>\n${historySummary}\n</previous-summary>\n\n`;
|
|
1073
|
-
}
|
|
1074
|
-
promptText += formatAdditionalContext(options?.extraContext);
|
|
1075
|
-
promptText += SHORT_SUMMARY_PROMPT;
|
|
1076
|
-
|
|
1077
|
-
if (options?.remoteEndpoint) {
|
|
1078
|
-
const remote = await requestRemoteCompaction(
|
|
1079
|
-
options.remoteEndpoint,
|
|
1080
|
-
{
|
|
1081
|
-
systemPrompt: SUMMARIZATION_SYSTEM_PROMPT,
|
|
1082
|
-
prompt: promptText,
|
|
1083
|
-
},
|
|
1084
|
-
signal,
|
|
1085
|
-
);
|
|
1086
|
-
return remote.summary;
|
|
1087
|
-
}
|
|
1088
|
-
|
|
1089
|
-
const response = await instrumentedCompleteSimple(
|
|
1090
|
-
model,
|
|
1091
|
-
{
|
|
1092
|
-
systemPrompt: [SUMMARIZATION_SYSTEM_PROMPT],
|
|
1093
|
-
messages: [{ role: "user", content: [{ type: "text", text: promptText }], timestamp: Date.now() }],
|
|
1094
|
-
},
|
|
1095
|
-
{
|
|
1096
|
-
maxTokens,
|
|
1097
|
-
signal,
|
|
1098
|
-
apiKey,
|
|
1099
|
-
reasoning: Effort.High,
|
|
1100
|
-
initiatorOverride: options?.initiatorOverride,
|
|
1101
|
-
metadata: options?.metadata,
|
|
1102
|
-
sessionId: options?.sessionId,
|
|
1103
|
-
providerSessionState: options?.providerSessionState,
|
|
1104
|
-
preferWebsockets: options?.preferWebsockets,
|
|
1105
|
-
},
|
|
1106
|
-
{ telemetry: options?.telemetry, oneshotKind: "compaction_short_summary" },
|
|
1107
|
-
);
|
|
1108
|
-
|
|
1109
|
-
if (response.stopReason === "error") {
|
|
1110
|
-
throw new Error(`Short summary failed: ${response.errorMessage || "Unknown error"}`);
|
|
1111
|
-
}
|
|
1112
|
-
|
|
1113
|
-
return response.content
|
|
1114
|
-
.filter((c): c is { type: "text"; text: string } => c.type === "text")
|
|
1115
|
-
.map(c => c.text)
|
|
1116
|
-
.join("\n");
|
|
1052
|
+
/** Derive a display summary locally to avoid a second compaction LLM request. */
|
|
1053
|
+
function deriveShortSummary(summary: string): string {
|
|
1054
|
+
const firstParagraph = summary.trim().split(/\n\s*\n/, 1)[0] ?? "";
|
|
1055
|
+
const maxLength = 2_000;
|
|
1056
|
+
return firstParagraph.length <= maxLength ? firstParagraph : `${firstParagraph.slice(0, maxLength - 1)}…`;
|
|
1117
1057
|
}
|
|
1118
1058
|
|
|
1119
1059
|
// ============================================================================
|
|
@@ -1163,6 +1103,11 @@ export interface PrepareCompactionOptions {
|
|
|
1163
1103
|
* (the confounded raw promptTokens/estimatedTokens quotient is never used).
|
|
1164
1104
|
*/
|
|
1165
1105
|
tokenCorrectionRatio?: number;
|
|
1106
|
+
/**
|
|
1107
|
+
* Model context-window size. Windows below 66k retain the legacy fixed
|
|
1108
|
+
* keepRecentTokens behavior; larger windows scale the keep window to 30%.
|
|
1109
|
+
*/
|
|
1110
|
+
contextWindow?: number;
|
|
1166
1111
|
}
|
|
1167
1112
|
|
|
1168
1113
|
export function prepareCompaction(
|
|
@@ -1193,13 +1138,42 @@ export function prepareCompaction(
|
|
|
1193
1138
|
// counts system+tools+full history while estimatedTokens counted only the
|
|
1194
1139
|
// post-boundary slice, so it was confounded and only ever shrank the window.
|
|
1195
1140
|
// Here the correction is bidirectional and clamped to [0.5, 2].
|
|
1196
|
-
const
|
|
1141
|
+
const configuredKeepRecentTokens = settings.keepRecentTokens;
|
|
1142
|
+
const contextWindow = options.contextWindow;
|
|
1143
|
+
const thresholdSafeKeepRecentTokens =
|
|
1144
|
+
contextWindow !== undefined && Number.isFinite(contextWindow) && contextWindow > 1
|
|
1145
|
+
? Math.max(
|
|
1146
|
+
1,
|
|
1147
|
+
resolveThresholdTokens(contextWindow, settings) - effectiveReserveTokens(contextWindow, settings, 0),
|
|
1148
|
+
)
|
|
1149
|
+
: configuredKeepRecentTokens;
|
|
1150
|
+
const keepRecentTokens = Math.min(configuredKeepRecentTokens, thresholdSafeKeepRecentTokens);
|
|
1151
|
+
// Preserve the legacy fixed window for smaller models. At 66k and above,
|
|
1152
|
+
// retain up to 30% of the model context, but never enough to leave the
|
|
1153
|
+
// post-compaction prompt immediately above its configured threshold.
|
|
1154
|
+
const scaledKeepRecentTokens =
|
|
1155
|
+
contextWindow !== undefined && Number.isFinite(contextWindow) && contextWindow >= 66_000
|
|
1156
|
+
? Math.min(thresholdSafeKeepRecentTokens, Math.max(keepRecentTokens, Math.floor(contextWindow * 0.3)))
|
|
1157
|
+
: keepRecentTokens;
|
|
1197
1158
|
const rawRatio = options.tokenCorrectionRatio;
|
|
1198
1159
|
const appliedRatio =
|
|
1199
1160
|
rawRatio !== undefined && Number.isFinite(rawRatio) && rawRatio > 0
|
|
1200
1161
|
? Math.min(TOKEN_CORRECTION_MAX_RATIO, Math.max(TOKEN_CORRECTION_MIN_RATIO, rawRatio))
|
|
1201
1162
|
: 1;
|
|
1202
|
-
|
|
1163
|
+
// Preserve an explicit keep floor that already covers the whole history: manual
|
|
1164
|
+
// and emergency callers rely on prepareCompaction returning undefined rather
|
|
1165
|
+
// than manufacturing a summary with no useful reduction. Otherwise, a scaled
|
|
1166
|
+
// window that exceeds a short history falls back to the threshold-safe floor.
|
|
1167
|
+
const historyTokens = pathEntries
|
|
1168
|
+
.slice(boundaryStart, boundaryEnd)
|
|
1169
|
+
.reduce((tokens, entry) => tokens + estimateEntryTokens(entry), 0);
|
|
1170
|
+
const effectiveKeepRecentTokens =
|
|
1171
|
+
configuredKeepRecentTokens > historyTokens
|
|
1172
|
+
? configuredKeepRecentTokens
|
|
1173
|
+
: scaledKeepRecentTokens > keepRecentTokens && scaledKeepRecentTokens > historyTokens
|
|
1174
|
+
? keepRecentTokens
|
|
1175
|
+
: scaledKeepRecentTokens;
|
|
1176
|
+
const keepRecentTokensCorrected = Math.max(1, Math.round(effectiveKeepRecentTokens / appliedRatio));
|
|
1203
1177
|
|
|
1204
1178
|
const cutPoint = findCutPoint(pathEntries, boundaryStart, boundaryEnd, keepRecentTokensCorrected);
|
|
1205
1179
|
|
|
@@ -1437,28 +1411,10 @@ export async function compact(
|
|
|
1437
1411
|
summary = "No prior history.";
|
|
1438
1412
|
}
|
|
1439
1413
|
|
|
1440
|
-
const shortSummary = await generateShortSummary(
|
|
1441
|
-
recentMessages,
|
|
1442
|
-
summary,
|
|
1443
|
-
model,
|
|
1444
|
-
settings.reserveTokens,
|
|
1445
|
-
apiKey,
|
|
1446
|
-
signal,
|
|
1447
|
-
{
|
|
1448
|
-
extraContext: options?.extraContext,
|
|
1449
|
-
remoteEndpoint: summaryOptions.remoteEndpoint,
|
|
1450
|
-
initiatorOverride: summaryOptions.initiatorOverride,
|
|
1451
|
-
metadata: summaryOptions.metadata,
|
|
1452
|
-
telemetry: summaryOptions.telemetry,
|
|
1453
|
-
sessionId: summaryOptions.sessionId,
|
|
1454
|
-
providerSessionState: summaryOptions.providerSessionState,
|
|
1455
|
-
preferWebsockets: summaryOptions.preferWebsockets,
|
|
1456
|
-
},
|
|
1457
|
-
);
|
|
1458
|
-
|
|
1459
1414
|
// Compute file lists and append to summary
|
|
1460
1415
|
const { readFiles, modifiedFiles } = computeFileLists(fileOps);
|
|
1461
1416
|
summary = upsertFileOperations(summary, readFiles, modifiedFiles);
|
|
1417
|
+
const shortSummary = deriveShortSummary(summary);
|
|
1462
1418
|
|
|
1463
1419
|
if (!firstKeptEntryId) {
|
|
1464
1420
|
throw new Error("First kept entry has no ID - session may need migration");
|
|
@@ -9,8 +9,9 @@
|
|
|
9
9
|
*/
|
|
10
10
|
|
|
11
11
|
import type { ToolCall, ToolResultMessage } from "@gajae-code/ai";
|
|
12
|
+
import { sanitizeText } from "@gajae-code/utils";
|
|
12
13
|
import type { AgentMessage } from "../types";
|
|
13
|
-
import { estimateEntryTokens } from "./compaction";
|
|
14
|
+
import { estimateEntryTokens, estimateTextTokensHeuristic } from "./compaction";
|
|
14
15
|
import type { SessionEntry, SessionMessageEntry } from "./entries";
|
|
15
16
|
|
|
16
17
|
export interface PruneConfig {
|
|
@@ -48,6 +49,7 @@ export interface PruneResult {
|
|
|
48
49
|
}
|
|
49
50
|
|
|
50
51
|
const DIGEST_NOTICE_TOKEN_CAP_MULTIPLIER = 1.25;
|
|
52
|
+
const ERROR_DIGEST_NOTICE_MIN_CHARS = 240;
|
|
51
53
|
|
|
52
54
|
function createGenericPrunedNotice(tokens: number): string {
|
|
53
55
|
return `[Output truncated - ${tokens} tokens]`;
|
|
@@ -66,6 +68,17 @@ function firstErrorLine(text: string): string | undefined {
|
|
|
66
68
|
?.trim();
|
|
67
69
|
}
|
|
68
70
|
|
|
71
|
+
function firstNonEmptyLine(text: string): string | undefined {
|
|
72
|
+
return text
|
|
73
|
+
.split(/\r?\n/)
|
|
74
|
+
.find(line => line.trim().length > 0)
|
|
75
|
+
?.trim();
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
function lastNonEmptyLine(text: string): string | undefined {
|
|
79
|
+
return text.trim().split(/\r?\n/).filter(Boolean).at(-1)?.trim();
|
|
80
|
+
}
|
|
81
|
+
|
|
69
82
|
function truncateField(value: string, maxLength: number): string {
|
|
70
83
|
if (value.length <= maxLength) return value;
|
|
71
84
|
if (maxLength <= 1) return "…";
|
|
@@ -74,7 +87,7 @@ function truncateField(value: string, maxLength: number): string {
|
|
|
74
87
|
|
|
75
88
|
function resultDigest(message: ToolResultMessage): string | undefined {
|
|
76
89
|
const toolName = message.toolName.toLowerCase();
|
|
77
|
-
const text = firstTextContent(message);
|
|
90
|
+
const text = sanitizeText(firstTextContent(message));
|
|
78
91
|
if (toolName === "bash") {
|
|
79
92
|
const details = message as { details?: { exitCode?: unknown } };
|
|
80
93
|
const exitCode =
|
|
@@ -99,7 +112,12 @@ function resultDigest(message: ToolResultMessage): string | undefined {
|
|
|
99
112
|
.join("; ") || "search digest unavailable"
|
|
100
113
|
);
|
|
101
114
|
}
|
|
102
|
-
return undefined;
|
|
115
|
+
if (message.isError !== true) return undefined;
|
|
116
|
+
if (text.trim().length === 0) return "error=tool result failed without text";
|
|
117
|
+
const error = firstErrorLine(text);
|
|
118
|
+
if (error) return `error=${error}`;
|
|
119
|
+
const summary = firstNonEmptyLine(text) ?? lastNonEmptyLine(text);
|
|
120
|
+
return summary ? `summary=${summary}` : undefined;
|
|
103
121
|
}
|
|
104
122
|
|
|
105
123
|
function createPrunedNotice(tokens: number, message?: ToolResultMessage): string {
|
|
@@ -110,7 +128,9 @@ function createPrunedNotice(tokens: number, message?: ToolResultMessage): string
|
|
|
110
128
|
const maxTokens = Math.max(genericTokens, Math.floor(genericTokens * DIGEST_NOTICE_TOKEN_CAP_MULTIPLIER));
|
|
111
129
|
const prefix = `[Output truncated - ${tokens} tokens; `;
|
|
112
130
|
const suffix = "]";
|
|
113
|
-
const
|
|
131
|
+
const digestChars = maxTokens * 4 - prefix.length - suffix.length;
|
|
132
|
+
const maxChars =
|
|
133
|
+
message?.isError === true ? Math.max(ERROR_DIGEST_NOTICE_MIN_CHARS, digestChars) : Math.max(0, digestChars);
|
|
114
134
|
return `${prefix}${truncateField(digest, maxChars)}${suffix}`;
|
|
115
135
|
}
|
|
116
136
|
|
|
@@ -122,8 +142,7 @@ function getToolResultMessage(entry: SessionEntry): ToolResultMessage | undefine
|
|
|
122
142
|
}
|
|
123
143
|
|
|
124
144
|
function estimatePrunedSavings(tokens: number, notice: string): number {
|
|
125
|
-
|
|
126
|
-
return Math.max(0, tokens - noticeTokens);
|
|
145
|
+
return tokens - estimateTextTokensHeuristic(notice);
|
|
127
146
|
}
|
|
128
147
|
|
|
129
148
|
export interface AssistantArgumentPruneResult {
|
|
@@ -267,6 +286,56 @@ function readBasePath(path: string): string {
|
|
|
267
286
|
return base;
|
|
268
287
|
}
|
|
269
288
|
|
|
289
|
+
type ReadLineRange = { start: number; end: number };
|
|
290
|
+
|
|
291
|
+
const DEFAULT_READ_LINE_LIMIT = 500;
|
|
292
|
+
|
|
293
|
+
/** Parse trailing read selectors using the read tool's actual bounded default. */
|
|
294
|
+
function readLineRanges(path: string): ReadLineRange[] {
|
|
295
|
+
let target = path;
|
|
296
|
+
let raw = false;
|
|
297
|
+
while (/:(?:raw|conflicts)$/.test(target)) {
|
|
298
|
+
raw ||= target.endsWith(":raw");
|
|
299
|
+
target = target.replace(/:(?:raw|conflicts)$/, "");
|
|
300
|
+
}
|
|
301
|
+
const match = target.match(/:(\d+(?:[-+]\d+)?(?:,\d+(?:[-+]\d+)?)*)$/);
|
|
302
|
+
if (!match) return raw ? [{ start: 1, end: Number.POSITIVE_INFINITY }] : [];
|
|
303
|
+
return match[1].split(",").flatMap(part => {
|
|
304
|
+
const range = part.match(/^(\d+)(?:([-+])(\d+))?$/);
|
|
305
|
+
if (!range) return [];
|
|
306
|
+
const start = Number(range[1]);
|
|
307
|
+
const end =
|
|
308
|
+
range[2] === "+"
|
|
309
|
+
? start + Number(range[3]) - 1
|
|
310
|
+
: range[2] === "-"
|
|
311
|
+
? Number(range[3])
|
|
312
|
+
: start + DEFAULT_READ_LINE_LIMIT - 1;
|
|
313
|
+
return start > 0 && end >= start ? [{ start, end }] : [];
|
|
314
|
+
});
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
function strictlyContainsReadRange(container: ReadLineRange, contained: ReadLineRange): boolean {
|
|
318
|
+
return (
|
|
319
|
+
container.start <= contained.start &&
|
|
320
|
+
container.end >= contained.end &&
|
|
321
|
+
(container.start < contained.start || container.end > contained.end)
|
|
322
|
+
);
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
function readSupersedesRead(
|
|
326
|
+
later: ToolCall,
|
|
327
|
+
earlier: ToolCall,
|
|
328
|
+
lineRangesByCall: ReadonlyMap<ToolCall, ReadLineRange[]>,
|
|
329
|
+
): boolean {
|
|
330
|
+
const laterRanges = lineRangesByCall.get(later);
|
|
331
|
+
const earlierRanges = lineRangesByCall.get(earlier);
|
|
332
|
+
return (
|
|
333
|
+
laterRanges?.length === 1 &&
|
|
334
|
+
earlierRanges?.length === 1 &&
|
|
335
|
+
strictlyContainsReadRange(laterRanges[0], earlierRanges[0])
|
|
336
|
+
);
|
|
337
|
+
}
|
|
338
|
+
|
|
270
339
|
/**
|
|
271
340
|
* Stable identity for "the same logical lookup": same tool re-targeting the
|
|
272
341
|
* same subject. A later result with the same key supersedes earlier ones.
|
|
@@ -275,9 +344,23 @@ function readBasePath(path: string): string {
|
|
|
275
344
|
* (`skip`) and result-shaping flags (`i`, `gitignore`): a later page or a
|
|
276
345
|
* differently-shaped search complements earlier output, it does not replace it.
|
|
277
346
|
*/
|
|
347
|
+
const IDEMPOTENT_BASH_COMMAND =
|
|
348
|
+
/^(?:(?:bun|npm|pnpm|yarn)\s+(?:run\s+)?(?:test|build)\b|git\s+status\b|cargo\s+build\b|(?:make|just)\s+build\b)/;
|
|
349
|
+
|
|
350
|
+
function normalizedIdempotentBashCommand(call: ToolCall): string | undefined {
|
|
351
|
+
if (call.name !== "bash") return undefined;
|
|
352
|
+
const command = call.arguments.command;
|
|
353
|
+
if (typeof command !== "string") return undefined;
|
|
354
|
+
const normalized = command.trim().replace(/\s+/g, " ");
|
|
355
|
+
if (/[;&|]/.test(normalized) || !IDEMPOTENT_BASH_COMMAND.test(normalized)) return undefined;
|
|
356
|
+
return JSON.stringify([normalized, typeof call.arguments.cwd === "string" ? call.arguments.cwd : undefined]);
|
|
357
|
+
}
|
|
358
|
+
|
|
278
359
|
function toolTargetKey(call: ToolCall): string | undefined {
|
|
279
360
|
const path = toolCallPath(call);
|
|
280
361
|
if (path !== undefined) return JSON.stringify([call.name, "path", path]);
|
|
362
|
+
const command = normalizedIdempotentBashCommand(call);
|
|
363
|
+
if (command !== undefined) return JSON.stringify([call.name, "command", command]);
|
|
281
364
|
const pattern = call.arguments.pattern;
|
|
282
365
|
if (typeof pattern === "string" && pattern.length > 0) {
|
|
283
366
|
const paths = call.arguments.paths;
|
|
@@ -372,8 +455,9 @@ function buildStalenessIndex(entries: SessionEntry[]): StalenessIndex {
|
|
|
372
455
|
}
|
|
373
456
|
}
|
|
374
457
|
|
|
458
|
+
type ResultMeta = { key?: string; call: ToolCall; message: ToolResultMessage };
|
|
375
459
|
const lastResultIndexByKey = new Map<string, number>();
|
|
376
|
-
const resultMeta = new Map<number,
|
|
460
|
+
const resultMeta = new Map<number, ResultMeta>();
|
|
377
461
|
const lastEditIndexByPath = new Map<string, number>();
|
|
378
462
|
|
|
379
463
|
for (let i = 0; i < entries.length; i++) {
|
|
@@ -439,6 +523,31 @@ function buildStalenessIndex(entries: SessionEntry[]): StalenessIndex {
|
|
|
439
523
|
}
|
|
440
524
|
}
|
|
441
525
|
|
|
526
|
+
const readsByBasePath = new Map<string, Array<[number, ResultMeta]>>();
|
|
527
|
+
const lineRangesByCall = new Map<ToolCall, ReadLineRange[]>();
|
|
528
|
+
for (const [index, meta] of resultMeta) {
|
|
529
|
+
if (meta.call.name !== "read") continue;
|
|
530
|
+
const path = toolCallPath(meta.call);
|
|
531
|
+
if (!path) continue;
|
|
532
|
+
lineRangesByCall.set(meta.call, readLineRanges(path));
|
|
533
|
+
const basePath = readBasePath(path);
|
|
534
|
+
const group = readsByBasePath.get(basePath);
|
|
535
|
+
if (group) group.push([index, meta]);
|
|
536
|
+
else readsByBasePath.set(basePath, [[index, meta]]);
|
|
537
|
+
}
|
|
538
|
+
for (const reads of readsByBasePath.values()) {
|
|
539
|
+
if (reads.length < 2) continue;
|
|
540
|
+
for (let earlier = 0; earlier < reads.length - 1; earlier++) {
|
|
541
|
+
const [index, meta] = reads[earlier];
|
|
542
|
+
for (let later = earlier + 1; later < reads.length; later++) {
|
|
543
|
+
if (readSupersedesRead(reads[later][1].call, meta.call, lineRangesByCall)) {
|
|
544
|
+
staleResultIndices.add(index);
|
|
545
|
+
break;
|
|
546
|
+
}
|
|
547
|
+
}
|
|
548
|
+
}
|
|
549
|
+
}
|
|
550
|
+
|
|
442
551
|
return { staleResultIndices };
|
|
443
552
|
}
|
|
444
553
|
export function pruneAssistantToolArguments(
|
|
@@ -580,11 +689,17 @@ function collectToolOutputPruneCandidates(
|
|
|
580
689
|
}
|
|
581
690
|
|
|
582
691
|
const notice = createPrunedNotice(tokens, message);
|
|
692
|
+
const savings = estimatePrunedSavings(tokens, notice);
|
|
693
|
+
const errorNoticeGrows = message.isError === true && notice.length > firstTextContent(message).length;
|
|
694
|
+
if (savings <= 0 || errorNoticeGrows) {
|
|
695
|
+
accumulatedTokens += tokens;
|
|
696
|
+
continue;
|
|
697
|
+
}
|
|
583
698
|
candidates.push({
|
|
584
699
|
entry: entry as SessionMessageEntry,
|
|
585
700
|
tokens,
|
|
586
701
|
notice,
|
|
587
|
-
savings
|
|
702
|
+
savings,
|
|
588
703
|
});
|
|
589
704
|
accumulatedTokens += tokens;
|
|
590
705
|
}
|
|
@@ -596,6 +711,13 @@ function collectToolOutputPruneCandidates(
|
|
|
596
711
|
return { candidates, tokensSaved };
|
|
597
712
|
}
|
|
598
713
|
|
|
714
|
+
function minimumSavings(config: PruneConfig, options: PruneToolOutputsOptions = {}): number {
|
|
715
|
+
const relaxedMinimum = options.relaxedMinimum;
|
|
716
|
+
return typeof relaxedMinimum === "number" && Number.isFinite(relaxedMinimum)
|
|
717
|
+
? Math.min(config.minimumSavings, Math.max(0, relaxedMinimum))
|
|
718
|
+
: config.minimumSavings;
|
|
719
|
+
}
|
|
720
|
+
|
|
599
721
|
/**
|
|
600
722
|
* Estimate the token savings {@link pruneToolOutputs} would achieve, without
|
|
601
723
|
* mutating any entry. Returns 0 savings when below the configured minimum so the
|
|
@@ -604,9 +726,10 @@ function collectToolOutputPruneCandidates(
|
|
|
604
726
|
export function estimateToolOutputPruneSavings(
|
|
605
727
|
entries: SessionEntry[],
|
|
606
728
|
config: PruneConfig = DEFAULT_PRUNE_CONFIG,
|
|
729
|
+
options: PruneToolOutputsOptions = {},
|
|
607
730
|
): { prunableCount: number; tokensSaved: number } {
|
|
608
731
|
const { candidates, tokensSaved } = collectToolOutputPruneCandidates(entries, config);
|
|
609
|
-
if (tokensSaved < config
|
|
732
|
+
if (tokensSaved < minimumSavings(config, options) || candidates.length === 0) {
|
|
610
733
|
return { prunableCount: 0, tokensSaved: 0 };
|
|
611
734
|
}
|
|
612
735
|
return { prunableCount: candidates.length, tokensSaved };
|
|
@@ -630,10 +753,20 @@ export function shouldRunMaintenancePrune(args: {
|
|
|
630
753
|
return args.estimatedSavings > args.cacheEpochResetCost;
|
|
631
754
|
}
|
|
632
755
|
|
|
633
|
-
export
|
|
756
|
+
export interface PruneToolOutputsOptions {
|
|
757
|
+
/** Lower the usual minimum only when the caller is already over its compaction threshold. */
|
|
758
|
+
relaxedMinimum?: number;
|
|
759
|
+
}
|
|
760
|
+
|
|
761
|
+
export function pruneToolOutputs(
|
|
762
|
+
entries: SessionEntry[],
|
|
763
|
+
config: PruneConfig = DEFAULT_PRUNE_CONFIG,
|
|
764
|
+
options: PruneToolOutputsOptions = {},
|
|
765
|
+
): PruneResult {
|
|
634
766
|
const { candidates, tokensSaved } = collectToolOutputPruneCandidates(entries, config);
|
|
767
|
+
const minimum = minimumSavings(config, options);
|
|
635
768
|
|
|
636
|
-
if (tokensSaved <
|
|
769
|
+
if (tokensSaved < minimum || candidates.length === 0) {
|
|
637
770
|
return { prunedCount: 0, tokensSaved: 0, prunedEntries: [] };
|
|
638
771
|
}
|
|
639
772
|
|
package/src/proxy.ts
CHANGED
|
@@ -30,6 +30,33 @@ class ProxyMessageEventStream extends EventStream<AssistantMessageEvent, Assista
|
|
|
30
30
|
}
|
|
31
31
|
}
|
|
32
32
|
|
|
33
|
+
interface ReasoningBuffers {
|
|
34
|
+
summary: string;
|
|
35
|
+
raw: string;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
const reasoningBuffers = new WeakMap<object, ReasoningBuffers>();
|
|
39
|
+
|
|
40
|
+
function materializeReasoningProvenance(
|
|
41
|
+
content: Extract<AssistantMessage["content"][number], { type: "thinking" }>,
|
|
42
|
+
): void {
|
|
43
|
+
const buffers = reasoningBuffers.get(content);
|
|
44
|
+
if (!buffers) return;
|
|
45
|
+
const mutable = content as { provenance?: "summary" | "raw" | "mixed"; summaryText?: string; rawText?: string };
|
|
46
|
+
if (mutable.provenance === undefined) {
|
|
47
|
+
if (mutable.summaryText === undefined && buffers.summary) mutable.summaryText = buffers.summary;
|
|
48
|
+
if (mutable.rawText === undefined && buffers.raw) mutable.rawText = buffers.raw;
|
|
49
|
+
mutable.provenance =
|
|
50
|
+
buffers.summary && buffers.raw ? "mixed" : buffers.summary ? "summary" : buffers.raw ? "raw" : undefined;
|
|
51
|
+
}
|
|
52
|
+
// Finalized display string must exclude raw CoT when a summary exists (parity with
|
|
53
|
+
// the Responses/Codex decoders): summary/mixed -> summary only; raw-only -> raw. The
|
|
54
|
+
// raw text stays available separately via rawText for explicit consumers.
|
|
55
|
+
const effSummary = mutable.summaryText ?? buffers.summary;
|
|
56
|
+
const effRaw = mutable.rawText ?? buffers.raw;
|
|
57
|
+
content.thinking = mutable.provenance === "raw" ? effRaw : effSummary || effRaw;
|
|
58
|
+
}
|
|
59
|
+
|
|
33
60
|
/**
|
|
34
61
|
* Proxy event types - server sends these with partial field stripped to reduce bandwidth.
|
|
35
62
|
*/
|
|
@@ -41,6 +68,9 @@ export type ProxyAssistantMessageEvent =
|
|
|
41
68
|
| { type: "thinking_start"; contentIndex: number }
|
|
42
69
|
| { type: "thinking_delta"; contentIndex: number; delta: string }
|
|
43
70
|
| { type: "thinking_end"; contentIndex: number; contentSignature?: string }
|
|
71
|
+
| { type: "reasoning_summary_start"; contentIndex: number }
|
|
72
|
+
| { type: "reasoning_summary_delta"; contentIndex: number; delta: string }
|
|
73
|
+
| { type: "reasoning_summary_end"; contentIndex: number; content?: string }
|
|
44
74
|
| { type: "toolcall_start"; contentIndex: number; id: string; toolName: string }
|
|
45
75
|
| { type: "toolcall_delta"; contentIndex: number; delta: string }
|
|
46
76
|
| { type: "toolcall_end"; contentIndex: number }
|
|
@@ -238,14 +268,22 @@ function processProxyEvent(
|
|
|
238
268
|
throw new Error("Received text_end for non-text content");
|
|
239
269
|
}
|
|
240
270
|
|
|
241
|
-
case "thinking_start":
|
|
242
|
-
|
|
271
|
+
case "thinking_start": {
|
|
272
|
+
const content = { type: "thinking", thinking: "" } as Extract<
|
|
273
|
+
AssistantMessage["content"][number],
|
|
274
|
+
{ type: "thinking" }
|
|
275
|
+
>;
|
|
276
|
+
partial.content[proxyEvent.contentIndex] = content;
|
|
277
|
+
reasoningBuffers.set(content, { summary: "", raw: "" });
|
|
243
278
|
return { type: "thinking_start", contentIndex: proxyEvent.contentIndex, partial };
|
|
279
|
+
}
|
|
244
280
|
|
|
245
281
|
case "thinking_delta": {
|
|
246
282
|
const content = partial.content[proxyEvent.contentIndex];
|
|
247
283
|
if (content?.type === "thinking") {
|
|
248
284
|
content.thinking += proxyEvent.delta;
|
|
285
|
+
const buffers = reasoningBuffers.get(content);
|
|
286
|
+
if (buffers) buffers.raw += proxyEvent.delta;
|
|
249
287
|
return {
|
|
250
288
|
type: "thinking_delta",
|
|
251
289
|
contentIndex: proxyEvent.contentIndex,
|
|
@@ -256,10 +294,52 @@ function processProxyEvent(
|
|
|
256
294
|
throw new Error("Received thinking_delta for non-thinking content");
|
|
257
295
|
}
|
|
258
296
|
|
|
297
|
+
case "reasoning_summary_start":
|
|
298
|
+
return { type: "reasoning_summary_start", contentIndex: proxyEvent.contentIndex, partial };
|
|
299
|
+
|
|
300
|
+
case "reasoning_summary_delta": {
|
|
301
|
+
const content = partial.content[proxyEvent.contentIndex];
|
|
302
|
+
if (content?.type === "thinking") {
|
|
303
|
+
content.thinking += proxyEvent.delta;
|
|
304
|
+
const buffers = reasoningBuffers.get(content);
|
|
305
|
+
if (buffers) buffers.summary += proxyEvent.delta;
|
|
306
|
+
return {
|
|
307
|
+
type: "reasoning_summary_delta",
|
|
308
|
+
contentIndex: proxyEvent.contentIndex,
|
|
309
|
+
delta: proxyEvent.delta,
|
|
310
|
+
partial,
|
|
311
|
+
};
|
|
312
|
+
}
|
|
313
|
+
throw new Error("Received reasoning_summary_delta for non-thinking content");
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
case "reasoning_summary_end": {
|
|
317
|
+
const content = partial.content[proxyEvent.contentIndex];
|
|
318
|
+
if (content?.type === "thinking") {
|
|
319
|
+
const buffers = reasoningBuffers.get(content);
|
|
320
|
+
// Final-only summaries arrive with the text on the end event and no summary
|
|
321
|
+
// deltas, so the accumulated buffer is empty. Prefer the end event's content
|
|
322
|
+
// so the summary survives the proxy and materializes as provenance "summary".
|
|
323
|
+
if (buffers && !buffers.summary.trim() && proxyEvent.content) {
|
|
324
|
+
buffers.summary = proxyEvent.content;
|
|
325
|
+
if (!content.thinking.trim()) content.thinking = proxyEvent.content;
|
|
326
|
+
}
|
|
327
|
+
materializeReasoningProvenance(content);
|
|
328
|
+
return {
|
|
329
|
+
type: "reasoning_summary_end",
|
|
330
|
+
contentIndex: proxyEvent.contentIndex,
|
|
331
|
+
content: buffers?.summary || proxyEvent.content || "",
|
|
332
|
+
partial,
|
|
333
|
+
};
|
|
334
|
+
}
|
|
335
|
+
throw new Error("Received reasoning_summary_end for non-thinking content");
|
|
336
|
+
}
|
|
337
|
+
|
|
259
338
|
case "thinking_end": {
|
|
260
339
|
const content = partial.content[proxyEvent.contentIndex];
|
|
261
340
|
if (content?.type === "thinking") {
|
|
262
341
|
content.thinkingSignature = proxyEvent.contentSignature;
|
|
342
|
+
materializeReasoningProvenance(content);
|
|
263
343
|
return {
|
|
264
344
|
type: "thinking_end",
|
|
265
345
|
contentIndex: proxyEvent.contentIndex,
|