@gajae-code/agent-core 0.4.4 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +26 -0
- package/dist/types/agent.d.ts +2 -0
- package/dist/types/append-only-context.d.ts +1 -0
- package/dist/types/compaction/compaction.d.ts +30 -4
- package/dist/types/compaction/entries.d.ts +6 -0
- package/dist/types/compaction/openai.d.ts +2 -0
- package/dist/types/compaction/pruning.d.ts +20 -1
- package/dist/types/types.d.ts +4 -0
- package/package.json +4 -4
- package/src/agent-loop.ts +4 -0
- package/src/agent.ts +23 -9
- package/src/append-only-context.ts +109 -35
- package/src/compaction/compaction.ts +123 -15
- package/src/compaction/entries.ts +6 -0
- package/src/compaction/openai.ts +15 -6
- package/src/compaction/pruning.ts +353 -14
- package/src/types.ts +3 -0
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,32 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [0.5.0] - 2026-06-13
|
|
6
|
+
|
|
7
|
+
### Fixed
|
|
8
|
+
|
|
9
|
+
- Fixed compaction cut-point selection when the newest retained context ends in an uncuttable tool result, so automatic compaction can keep the latest assistant/tool-result pair instead of falling back to a no-op cut.
|
|
10
|
+
|
|
11
|
+
### Changed
|
|
12
|
+
|
|
13
|
+
- Optimization Suite v3 Lane 2 (context cost): compaction token estimates now use a shared per-entry cache (`estimateEntryTokens`) keyed by a boundary-lossless fingerprint of the exact estimator fragments, covering `estimateEntriesTokens`, the `findCutPoint` reverse walk, and pruning candidate scoring — repeated full-session estimate p95 −97%, token totals exactly equal to fresh estimates and never stale after prune mutation. Pruned bash/search/grep tool results now carry a one-line digest notice (exit code, match/file count, first error line; capped at 1.25× the generic notice cost) instead of a bare truncation notice, with savings computed from the exact notice string; reads and other tools keep the generic notice. `trimOpenAiCompactInput` is O(n) via per-item serialized lengths and a running character sum (5k-item trim −99.9%) and is now exported.
|
|
14
|
+
- Optimization Suite v3 Lane 3 (serialization): `cloneJson` in the append-only context now uses a typed JSON-semantic recursive clone instead of a `JSON.parse(JSON.stringify())` round-trip (−36% median on clone-heavy paths), with exact JSON.stringify byte parity including the toJSON holder-key protocol (single get, no re-dispatch on replacement values), function/symbol dropping, sparse arrays, Dates, and prototype-bearing objects; the helper is now exported.
|
|
15
|
+
|
|
16
|
+
## [0.4.5] - 2026-06-12
|
|
17
|
+
|
|
18
|
+
### Changed
|
|
19
|
+
|
|
20
|
+
- Made tool-output pruning staleness-aware: results superseded by a later same-target result (re-read file, re-run search) or invalidated by a later successful edit/write are pruned before merely-old ones, including inside the recency protect window. New optional `PruneConfig.staleOverridableTools` (default `["read"]`) waives protected-tool immunity for superseded results while the most recent result per target stays protected. Target identity uses collision-proof canonical JSON tuple keys.
|
|
21
|
+
- `PruneResult` now returns `prunedEntries` so callers whose entry source materializes copies (e.g. blob-externalized session entries) can write mutations back into their canonical store.
|
|
22
|
+
|
|
23
|
+
### Fixed
|
|
24
|
+
|
|
25
|
+
- Preserved Cursor-native tool call rendering and execution through the agent tool-call path, including runtime tool details.
|
|
26
|
+
|
|
27
|
+
## [0.4.4] - 2026-06-10
|
|
28
|
+
|
|
29
|
+
- Version aligned with the 0.4.4 monorepo release; no functional changes in this package.
|
|
30
|
+
|
|
5
31
|
## [0.4.3] - 2026-06-10
|
|
6
32
|
|
|
7
33
|
### Fixed
|
package/dist/types/agent.d.ts
CHANGED
|
@@ -80,6 +80,8 @@ export interface AgentOptions {
|
|
|
80
80
|
* Use this when abort decisions must happen before buffered events continue flowing.
|
|
81
81
|
*/
|
|
82
82
|
onAssistantMessageEvent?: (message: AssistantMessage, event: AssistantMessageEvent) => void;
|
|
83
|
+
/** Called for non-content tool-choice incapability stream events. */
|
|
84
|
+
onToolChoiceIncapability?: AgentLoopConfig["onToolChoiceIncapability"];
|
|
83
85
|
/**
|
|
84
86
|
* Called when GPT-5 Harmony protocol leakage is detected and mitigated.
|
|
85
87
|
*/
|
|
@@ -68,11 +68,37 @@ export declare function effectiveReserveTokens(contextWindow: number, settings:
|
|
|
68
68
|
export declare function shouldCompact(contextTokens: number, contextWindow: number, settings: CompactionSettings, maxOutputTokens?: number): boolean;
|
|
69
69
|
export declare function resolveThresholdTokens(contextWindow: number, settings: CompactionSettings, maxOutputTokens?: number): number;
|
|
70
70
|
/**
|
|
71
|
-
* Estimate token count for a message using
|
|
72
|
-
*
|
|
73
|
-
* publish
|
|
71
|
+
* Estimate token count for a message using the native o200k tokenizer.
|
|
72
|
+
* Exact for o200k only; an approximation for Anthropic/other model families
|
|
73
|
+
* (Anthropic doesn't publish a tokenizer) within ~5–10% on English/code text.
|
|
74
|
+
*
|
|
75
|
+
* This materializes the native BPE table (~50MB RSS) on first call. Use it
|
|
76
|
+
* only for context-changing decisions (compaction trigger/cut points, pruning
|
|
77
|
+
* budgets, branch summarization, fork-context seeding, context-limit
|
|
78
|
+
* enforcement). For display-only totals use
|
|
79
|
+
* {@link estimateMessageTokensHeuristic}.
|
|
80
|
+
*/
|
|
81
|
+
export declare function countMessageTokensNativeO200k(message: AgentMessage): number;
|
|
82
|
+
/**
|
|
83
|
+
* Backwards-compatible alias for {@link countMessageTokensNativeO200k}.
|
|
84
|
+
* Existing callers treat this as the canonical message-token estimator for
|
|
85
|
+
* context-changing decisions.
|
|
86
|
+
*/
|
|
87
|
+
export declare const estimateTokens: typeof countMessageTokensNativeO200k;
|
|
88
|
+
/**
|
|
89
|
+
* Cheap, native-free token estimate for a message. Suitable ONLY for
|
|
90
|
+
* display/init surfaces (status line, /context report, HUD totals) — never
|
|
91
|
+
* for context-changing decisions, which must use
|
|
92
|
+
* {@link countMessageTokensNativeO200k}.
|
|
93
|
+
*/
|
|
94
|
+
export declare function estimateMessageTokensHeuristic(message: AgentMessage): number;
|
|
95
|
+
/**
|
|
96
|
+
* Cheap, native-free token estimate for plain string fragments. Display-only
|
|
97
|
+
* counterpart of the native `countTokens(fragments)` aggregate.
|
|
74
98
|
*/
|
|
75
|
-
export declare function
|
|
99
|
+
export declare function estimateTextTokensHeuristic(fragments: string | readonly string[]): number;
|
|
100
|
+
export declare function estimateEntryTokens(entry: SessionEntry): number;
|
|
101
|
+
export declare function estimateEntriesTokens(entries: SessionEntry[], startIndex: number, endIndex: number): number;
|
|
76
102
|
/**
|
|
77
103
|
* Find the user message (or bashExecution) that starts the turn containing the given entry index.
|
|
78
104
|
* Returns -1 if no turn start found before the index.
|
|
@@ -20,6 +20,12 @@ export interface ModelChangeEntry extends SessionEntryBase {
|
|
|
20
20
|
model: string;
|
|
21
21
|
/** Role: "default", "smol", "slow", etc. Undefined treated as "default" */
|
|
22
22
|
role?: string;
|
|
23
|
+
/** Requested model before a runtime substitution/fallback, in "provider/modelId" format. */
|
|
24
|
+
previousModel?: string;
|
|
25
|
+
/** Machine-readable reason for runtime model substitution/fallback. */
|
|
26
|
+
reason?: string;
|
|
27
|
+
/** Effective thinking level when the change was recorded. */
|
|
28
|
+
thinkingLevel?: string | null;
|
|
23
29
|
}
|
|
24
30
|
export interface ServiceTierChangeEntry extends SessionEntryBase {
|
|
25
31
|
type: "service_tier_change";
|
|
@@ -41,6 +41,8 @@ export interface RemoteCompactionResponse {
|
|
|
41
41
|
export declare function shouldUseOpenAiRemoteCompaction(model: Model): boolean;
|
|
42
42
|
export declare function getPreservedOpenAiRemoteCompactionData(preserveData: Record<string, unknown> | undefined): OpenAiRemoteCompactionPreserveData | undefined;
|
|
43
43
|
export declare function withOpenAiRemoteCompactionPreserveData(preserveData: Record<string, unknown> | undefined, remoteCompaction: OpenAiRemoteCompactionPreserveData | undefined): Record<string, unknown> | undefined;
|
|
44
|
+
export declare function estimateOpenAiCompactInputTokens(input: Array<Record<string, unknown>>, instructions: string): number;
|
|
45
|
+
export declare function trimOpenAiCompactInput(input: Array<Record<string, unknown>>, contextWindow: number, instructions: string): Array<Record<string, unknown>>;
|
|
44
46
|
/**
|
|
45
47
|
* Build the OpenAI Responses-API native history array from LLM messages.
|
|
46
48
|
*
|
|
@@ -1,7 +1,13 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Tool output pruning utilities for compaction.
|
|
3
|
+
*
|
|
4
|
+
* Candidate selection is staleness-aware: tool results that have been
|
|
5
|
+
* superseded by a later result for the same target (same file read again,
|
|
6
|
+
* same search re-run) or invalidated by a later successful edit/write to a
|
|
7
|
+
* covered file are pruned in preference to merely-old results. Protect-window
|
|
8
|
+
* and minimum-savings hysteresis semantics are unchanged.
|
|
3
9
|
*/
|
|
4
|
-
import type { SessionEntry } from "./entries";
|
|
10
|
+
import type { SessionEntry, SessionMessageEntry } from "./entries";
|
|
5
11
|
export interface PruneConfig {
|
|
6
12
|
/** Keep the most recent tool output tokens intact. */
|
|
7
13
|
protectTokens: number;
|
|
@@ -9,10 +15,23 @@ export interface PruneConfig {
|
|
|
9
15
|
minimumSavings: number;
|
|
10
16
|
/** Tool names that should never be pruned. */
|
|
11
17
|
protectedTools: string[];
|
|
18
|
+
/**
|
|
19
|
+
* Tools in `protectedTools` whose protection is waived once the result is
|
|
20
|
+
* superseded (a later result for the same target, or a later successful
|
|
21
|
+
* edit/write to the covered file). The most recent result per target is
|
|
22
|
+
* never considered superseded. Optional; defaults to none.
|
|
23
|
+
*/
|
|
24
|
+
staleOverridableTools?: string[];
|
|
12
25
|
}
|
|
13
26
|
export declare const DEFAULT_PRUNE_CONFIG: PruneConfig;
|
|
14
27
|
export interface PruneResult {
|
|
15
28
|
prunedCount: number;
|
|
16
29
|
tokensSaved: number;
|
|
30
|
+
/**
|
|
31
|
+
* The mutated message entries. Callers whose entry source returns
|
|
32
|
+
* materialized copies (not live references) must write these back into
|
|
33
|
+
* their canonical store by id.
|
|
34
|
+
*/
|
|
35
|
+
prunedEntries: SessionMessageEntry[];
|
|
17
36
|
}
|
|
18
37
|
export declare function pruneToolOutputs(entries: SessionEntry[], config?: PruneConfig): PruneResult;
|
package/dist/types/types.d.ts
CHANGED
|
@@ -154,6 +154,10 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
|
|
|
154
154
|
* Callers may abort synchronously to stop consuming buffered provider events.
|
|
155
155
|
*/
|
|
156
156
|
onAssistantMessageEvent?: (message: AssistantMessage, event: AssistantMessageEvent) => void;
|
|
157
|
+
/** Called for non-content tool-choice incapability stream events. */
|
|
158
|
+
onToolChoiceIncapability?: (event: Extract<AssistantMessageEvent, {
|
|
159
|
+
type: "toolChoiceIncapability";
|
|
160
|
+
}>) => void;
|
|
157
161
|
/**
|
|
158
162
|
* Called when GPT-5 Harmony protocol leakage is detected and mitigated.
|
|
159
163
|
*/
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@gajae-code/agent-core",
|
|
4
|
-
"version": "0.
|
|
4
|
+
"version": "0.5.0",
|
|
5
5
|
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
|
|
6
6
|
"homepage": "https://gaebal-gajae.dev",
|
|
7
7
|
"author": "Yeachan-Heo",
|
|
@@ -35,9 +35,9 @@
|
|
|
35
35
|
"fmt": "biome format --write ."
|
|
36
36
|
},
|
|
37
37
|
"dependencies": {
|
|
38
|
-
"@gajae-code/ai": "0.
|
|
39
|
-
"@gajae-code/natives": "0.
|
|
40
|
-
"@gajae-code/utils": "0.
|
|
38
|
+
"@gajae-code/ai": "0.5.0",
|
|
39
|
+
"@gajae-code/natives": "0.5.0",
|
|
40
|
+
"@gajae-code/utils": "0.5.0",
|
|
41
41
|
"@opentelemetry/api": "^1.9.0"
|
|
42
42
|
},
|
|
43
43
|
"devDependencies": {
|
package/src/agent-loop.ts
CHANGED
|
@@ -815,6 +815,10 @@ async function streamAssistantResponse(
|
|
|
815
815
|
stream.push({ type: "message_start", message: { ...partialMessage } });
|
|
816
816
|
break;
|
|
817
817
|
|
|
818
|
+
case "toolChoiceIncapability":
|
|
819
|
+
config.onToolChoiceIncapability?.(event);
|
|
820
|
+
break;
|
|
821
|
+
|
|
818
822
|
case "text_start":
|
|
819
823
|
case "text_delta":
|
|
820
824
|
case "text_end":
|
package/src/agent.ts
CHANGED
|
@@ -151,6 +151,8 @@ export interface AgentOptions {
|
|
|
151
151
|
* Use this when abort decisions must happen before buffered events continue flowing.
|
|
152
152
|
*/
|
|
153
153
|
onAssistantMessageEvent?: (message: AssistantMessage, event: AssistantMessageEvent) => void;
|
|
154
|
+
/** Called for non-content tool-choice incapability stream events. */
|
|
155
|
+
onToolChoiceIncapability?: AgentLoopConfig["onToolChoiceIncapability"];
|
|
154
156
|
|
|
155
157
|
/**
|
|
156
158
|
* Called when GPT-5 Harmony protocol leakage is detected and mitigated.
|
|
@@ -309,6 +311,7 @@ export class Agent {
|
|
|
309
311
|
#onResponse?: SimpleStreamOptions["onResponse"];
|
|
310
312
|
#onSseEvent?: SimpleStreamOptions["onSseEvent"];
|
|
311
313
|
#onAssistantMessageEvent?: (message: AssistantMessage, event: AssistantMessageEvent) => void;
|
|
314
|
+
#onToolChoiceIncapability?: AgentLoopConfig["onToolChoiceIncapability"];
|
|
312
315
|
#onHarmonyLeak?: (event: HarmonyAuditEvent) => void | Promise<void>;
|
|
313
316
|
#onBeforeYield?: () => Promise<void> | void;
|
|
314
317
|
#shouldPause?: AgentLoopConfig["shouldPause"];
|
|
@@ -373,6 +376,7 @@ export class Agent {
|
|
|
373
376
|
this.#intentTracing = opts.intentTracing === true;
|
|
374
377
|
this.#getToolChoice = opts.getToolChoice;
|
|
375
378
|
this.#onAssistantMessageEvent = opts.onAssistantMessageEvent;
|
|
379
|
+
this.#onToolChoiceIncapability = opts.onToolChoiceIncapability;
|
|
376
380
|
this.#onHarmonyLeak = opts.onHarmonyLeak;
|
|
377
381
|
this.#shouldPause = opts.shouldPause;
|
|
378
382
|
this.beforeToolCall = opts.beforeToolCall;
|
|
@@ -680,7 +684,11 @@ export class Agent {
|
|
|
680
684
|
if (!source) return undefined;
|
|
681
685
|
|
|
682
686
|
const guarded: CursorExecHandlers = {};
|
|
683
|
-
|
|
687
|
+
// Bind each handler to `source`: they are methods of a CursorExecHandlers
|
|
688
|
+
// instance that reference private fields via `this`. Extracting them bare
|
|
689
|
+
// (`const read = source.read`) and calling `read(args)` would invoke them with
|
|
690
|
+
// `this === undefined`, throwing "undefined is not an object (this.#optionsForCall)".
|
|
691
|
+
const read = source.read?.bind(source);
|
|
684
692
|
if (read) {
|
|
685
693
|
guarded.read = async args => {
|
|
686
694
|
this.#assertActiveRun(runId);
|
|
@@ -689,7 +697,7 @@ export class Agent {
|
|
|
689
697
|
return result;
|
|
690
698
|
};
|
|
691
699
|
}
|
|
692
|
-
const ls = source.ls;
|
|
700
|
+
const ls = source.ls?.bind(source);
|
|
693
701
|
if (ls) {
|
|
694
702
|
guarded.ls = async args => {
|
|
695
703
|
this.#assertActiveRun(runId);
|
|
@@ -698,7 +706,7 @@ export class Agent {
|
|
|
698
706
|
return result;
|
|
699
707
|
};
|
|
700
708
|
}
|
|
701
|
-
const grep = source.grep;
|
|
709
|
+
const grep = source.grep?.bind(source);
|
|
702
710
|
if (grep) {
|
|
703
711
|
guarded.grep = async args => {
|
|
704
712
|
this.#assertActiveRun(runId);
|
|
@@ -707,7 +715,7 @@ export class Agent {
|
|
|
707
715
|
return result;
|
|
708
716
|
};
|
|
709
717
|
}
|
|
710
|
-
const write = source.write;
|
|
718
|
+
const write = source.write?.bind(source);
|
|
711
719
|
if (write) {
|
|
712
720
|
guarded.write = async args => {
|
|
713
721
|
this.#assertActiveRun(runId);
|
|
@@ -716,7 +724,7 @@ export class Agent {
|
|
|
716
724
|
return result;
|
|
717
725
|
};
|
|
718
726
|
}
|
|
719
|
-
const deleteHandler = source.delete;
|
|
727
|
+
const deleteHandler = source.delete?.bind(source);
|
|
720
728
|
if (deleteHandler) {
|
|
721
729
|
guarded.delete = async args => {
|
|
722
730
|
this.#assertActiveRun(runId);
|
|
@@ -725,7 +733,7 @@ export class Agent {
|
|
|
725
733
|
return result;
|
|
726
734
|
};
|
|
727
735
|
}
|
|
728
|
-
const shell = source.shell;
|
|
736
|
+
const shell = source.shell?.bind(source);
|
|
729
737
|
if (shell) {
|
|
730
738
|
guarded.shell = async args => {
|
|
731
739
|
this.#assertActiveRun(runId);
|
|
@@ -734,7 +742,7 @@ export class Agent {
|
|
|
734
742
|
return result;
|
|
735
743
|
};
|
|
736
744
|
}
|
|
737
|
-
const shellStream = source.shellStream;
|
|
745
|
+
const shellStream = source.shellStream?.bind(source);
|
|
738
746
|
if (shellStream) {
|
|
739
747
|
guarded.shellStream = async (args, callbacks) => {
|
|
740
748
|
this.#assertActiveRun(runId);
|
|
@@ -743,7 +751,7 @@ export class Agent {
|
|
|
743
751
|
return result;
|
|
744
752
|
};
|
|
745
753
|
}
|
|
746
|
-
const diagnostics = source.diagnostics;
|
|
754
|
+
const diagnostics = source.diagnostics?.bind(source);
|
|
747
755
|
if (diagnostics) {
|
|
748
756
|
guarded.diagnostics = async args => {
|
|
749
757
|
this.#assertActiveRun(runId);
|
|
@@ -752,7 +760,7 @@ export class Agent {
|
|
|
752
760
|
return result;
|
|
753
761
|
};
|
|
754
762
|
}
|
|
755
|
-
const mcp = source.mcp;
|
|
763
|
+
const mcp = source.mcp?.bind(source);
|
|
756
764
|
if (mcp) {
|
|
757
765
|
guarded.mcp = async call => {
|
|
758
766
|
this.#assertActiveRun(runId);
|
|
@@ -1218,6 +1226,12 @@ export class Agent {
|
|
|
1218
1226
|
this.#onAssistantMessageEvent?.(message, event);
|
|
1219
1227
|
}
|
|
1220
1228
|
: undefined,
|
|
1229
|
+
onToolChoiceIncapability: this.#onToolChoiceIncapability
|
|
1230
|
+
? event => {
|
|
1231
|
+
if (this.#activeRunId !== runId) return;
|
|
1232
|
+
this.#onToolChoiceIncapability?.(event);
|
|
1233
|
+
}
|
|
1234
|
+
: undefined,
|
|
1221
1235
|
onHarmonyLeak: this.#onHarmonyLeak,
|
|
1222
1236
|
getToolChoice,
|
|
1223
1237
|
getReasoning: () => this.#state.thinkingLevel,
|
|
@@ -46,6 +46,9 @@ export interface BuildOptions {
|
|
|
46
46
|
export class StablePrefix {
|
|
47
47
|
#snapshot: StablePrefixSnapshot | null = null;
|
|
48
48
|
#version = 0;
|
|
49
|
+
#sourceSystemPrompt: readonly string[] | null = null;
|
|
50
|
+
#sourceTools: AgentContext["tools"] | null = null;
|
|
51
|
+
#sourceIntentTracing: boolean | null = null;
|
|
49
52
|
|
|
50
53
|
get fingerprint(): string {
|
|
51
54
|
return this.#snapshot?.fingerprint ?? "<unbuilt>";
|
|
@@ -65,6 +68,9 @@ export class StablePrefix {
|
|
|
65
68
|
const systemPrompt = cloneJson(snapshot.systemPrompt);
|
|
66
69
|
const tools = normalizeImportedTools(snapshot.tools, options);
|
|
67
70
|
const fingerprint = computeFingerprint(systemPrompt, tools, options);
|
|
71
|
+
this.#sourceSystemPrompt = null;
|
|
72
|
+
this.#sourceTools = null;
|
|
73
|
+
this.#sourceIntentTracing = null;
|
|
68
74
|
if (fingerprint !== snapshot.fingerprint) {
|
|
69
75
|
throw new Error(
|
|
70
76
|
`StablePrefix.importSnapshot() fingerprint mismatch: expected ${fingerprint}, received ${snapshot.fingerprint}`,
|
|
@@ -79,11 +85,26 @@ export class StablePrefix {
|
|
|
79
85
|
* Returns `true` if the prefix actually changed (cache miss imminent).
|
|
80
86
|
*/
|
|
81
87
|
build(context: AgentContext, options: BuildOptions): boolean {
|
|
88
|
+
if (
|
|
89
|
+
this.#snapshot &&
|
|
90
|
+
this.#sourceSystemPrompt === context.systemPrompt &&
|
|
91
|
+
this.#sourceTools === context.tools &&
|
|
92
|
+
this.#sourceIntentTracing === options.intentTracing
|
|
93
|
+
) {
|
|
94
|
+
const sourceFingerprint = takeSnapshot(context, options).fingerprint;
|
|
95
|
+
if (this.#snapshot.fingerprint === sourceFingerprint) return false;
|
|
96
|
+
}
|
|
82
97
|
const snapshot = takeSnapshot(context, options);
|
|
83
98
|
if (this.#snapshot && this.#snapshot.fingerprint === snapshot.fingerprint) {
|
|
99
|
+
this.#sourceSystemPrompt = context.systemPrompt;
|
|
100
|
+
this.#sourceTools = context.tools;
|
|
101
|
+
this.#sourceIntentTracing = options.intentTracing;
|
|
84
102
|
return false;
|
|
85
103
|
}
|
|
86
104
|
this.#snapshot = snapshot;
|
|
105
|
+
this.#sourceSystemPrompt = context.systemPrompt;
|
|
106
|
+
this.#sourceTools = context.tools;
|
|
107
|
+
this.#sourceIntentTracing = options.intentTracing;
|
|
87
108
|
this.#version++;
|
|
88
109
|
return true;
|
|
89
110
|
}
|
|
@@ -91,6 +112,9 @@ export class StablePrefix {
|
|
|
91
112
|
/** Force rebuild on the next `build()` call. */
|
|
92
113
|
invalidate(): void {
|
|
93
114
|
this.#snapshot = null;
|
|
115
|
+
this.#sourceSystemPrompt = null;
|
|
116
|
+
this.#sourceTools = null;
|
|
117
|
+
this.#sourceIntentTracing = null;
|
|
94
118
|
}
|
|
95
119
|
|
|
96
120
|
/**
|
|
@@ -175,8 +199,8 @@ export class AppendOnlyContextManager {
|
|
|
175
199
|
readonly log = new AppendOnlyLog();
|
|
176
200
|
/** How many normalized messages were synced into the log as of the last sync. */
|
|
177
201
|
#lastSyncCount = 0;
|
|
178
|
-
/**
|
|
179
|
-
#syncedDigest =
|
|
202
|
+
/** Fingerprint plus source bytes of synced message content — detects in-place rewrites with no hash-only equality. */
|
|
203
|
+
#syncedDigest = emptyMessageDigest();
|
|
180
204
|
/** Number of provider-normalized messages that were seeded before child-local messages. */
|
|
181
205
|
#seededPrefixCount = 0;
|
|
182
206
|
|
|
@@ -208,19 +232,22 @@ export class AppendOnlyContextManager {
|
|
|
208
232
|
* (same length, changed content via a rolling digest).
|
|
209
233
|
*/
|
|
210
234
|
syncMessages(normalizedMessages: any[]): void {
|
|
211
|
-
const
|
|
235
|
+
const seededPrefixLength = this.#seededPrefixCount;
|
|
212
236
|
const includesSeedPrefix =
|
|
213
|
-
|
|
214
|
-
normalizedMessages.length >=
|
|
215
|
-
this.#
|
|
237
|
+
seededPrefixLength > 0 &&
|
|
238
|
+
normalizedMessages.length >= seededPrefixLength &&
|
|
239
|
+
this.#computeDigestRange(normalizedMessages, 0, seededPrefixLength).source ===
|
|
240
|
+
this.#computeDigestRange(this.log.entries(), 0, seededPrefixLength).source;
|
|
216
241
|
const messagesToSync =
|
|
217
|
-
|
|
242
|
+
seededPrefixLength > 0 && !includesSeedPrefix
|
|
243
|
+
? [...this.log.entries().slice(0, seededPrefixLength), ...normalizedMessages]
|
|
244
|
+
: normalizedMessages;
|
|
218
245
|
|
|
219
246
|
// Detect in-place rewrites of already-synced messages.
|
|
220
247
|
if (
|
|
221
248
|
this.#lastSyncCount > 0 &&
|
|
222
249
|
this.#lastSyncCount <= messagesToSync.length &&
|
|
223
|
-
this.#
|
|
250
|
+
this.#computeDigestRange(messagesToSync, 0, this.#lastSyncCount).source !== this.#syncedDigest.source
|
|
224
251
|
) {
|
|
225
252
|
if (this.#seededPrefixCount > 0) {
|
|
226
253
|
throw new Error("AppendOnlyContextManager.syncMessages() seed prefix changed");
|
|
@@ -266,7 +293,7 @@ export class AppendOnlyContextManager {
|
|
|
266
293
|
this.prefix.invalidate();
|
|
267
294
|
this.log.clear();
|
|
268
295
|
this.#lastSyncCount = 0;
|
|
269
|
-
this.#syncedDigest =
|
|
296
|
+
this.#syncedDigest = emptyMessageDigest();
|
|
270
297
|
this.#seededPrefixCount = 0;
|
|
271
298
|
}
|
|
272
299
|
|
|
@@ -274,7 +301,7 @@ export class AppendOnlyContextManager {
|
|
|
274
301
|
resetSyncCursor(): void {
|
|
275
302
|
this.log.clear();
|
|
276
303
|
this.#lastSyncCount = 0;
|
|
277
|
-
this.#syncedDigest =
|
|
304
|
+
this.#syncedDigest = emptyMessageDigest();
|
|
278
305
|
this.#seededPrefixCount = 0;
|
|
279
306
|
}
|
|
280
307
|
|
|
@@ -294,36 +321,28 @@ export class AppendOnlyContextManager {
|
|
|
294
321
|
this.prefix.invalidate();
|
|
295
322
|
this.log.clear();
|
|
296
323
|
this.#lastSyncCount = 0;
|
|
297
|
-
this.#syncedDigest =
|
|
324
|
+
this.#syncedDigest = emptyMessageDigest();
|
|
298
325
|
this.#seededPrefixCount = 0;
|
|
299
326
|
this.prefix.build(context, options);
|
|
300
327
|
}
|
|
301
328
|
|
|
302
329
|
/**
|
|
303
|
-
* Deterministic digest over
|
|
304
|
-
*
|
|
305
|
-
*
|
|
306
|
-
* accumulator so in-place rewrites of *any* of these fields are visible.
|
|
330
|
+
* Deterministic digest over the provider-visible message payload. The source
|
|
331
|
+
* string is kept and compared for equality so the hash is only a fast summary,
|
|
332
|
+
* never the authority for accepting append-only sync state.
|
|
307
333
|
*/
|
|
308
|
-
#computeDigest(messages: readonly unknown[]):
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
tc: m.toolCalls ?? m.tool_calls ?? null,
|
|
318
|
-
tcid: m.tool_call_id ?? null,
|
|
319
|
-
n: m.name ?? null,
|
|
320
|
-
id: m.id ?? null,
|
|
321
|
-
});
|
|
322
|
-
for (let j = 0; j < payload.length; j++) {
|
|
323
|
-
hash = ((hash << 5) - hash + payload.charCodeAt(j)) | 0;
|
|
324
|
-
}
|
|
334
|
+
#computeDigest(messages: readonly unknown[]): MessageDigest {
|
|
335
|
+
return this.#computeDigestRange(messages, 0, messages.length);
|
|
336
|
+
}
|
|
337
|
+
|
|
338
|
+
#computeDigestRange(messages: readonly unknown[], start: number, end: number): MessageDigest {
|
|
339
|
+
let source = "[";
|
|
340
|
+
for (let i = start; i < end; i++) {
|
|
341
|
+
if (i > start) source += ",";
|
|
342
|
+
source += JSON.stringify(messages[i]) ?? "null";
|
|
325
343
|
}
|
|
326
|
-
|
|
344
|
+
source += "]";
|
|
345
|
+
return { hash: hashSource(source), source };
|
|
327
346
|
}
|
|
328
347
|
}
|
|
329
348
|
|
|
@@ -331,6 +350,27 @@ export class AppendOnlyContextManager {
|
|
|
331
350
|
// Snapshot helpers
|
|
332
351
|
// ---------------------------------------------------------------------------
|
|
333
352
|
|
|
353
|
+
type MessageDigest = {
|
|
354
|
+
hash: number | bigint;
|
|
355
|
+
source: string;
|
|
356
|
+
};
|
|
357
|
+
|
|
358
|
+
function emptyMessageDigest(): MessageDigest {
|
|
359
|
+
return { hash: hashSource("[]"), source: "[]" };
|
|
360
|
+
}
|
|
361
|
+
|
|
362
|
+
function hashSource(source: string): number | bigint {
|
|
363
|
+
return typeof Bun !== "undefined" ? Bun.hash(source) : hashString32(source);
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
function hashString32(value: string): number {
|
|
367
|
+
let hash = 0;
|
|
368
|
+
for (let i = 0; i < value.length; i++) {
|
|
369
|
+
hash = ((hash << 5) - hash + value.charCodeAt(i)) | 0;
|
|
370
|
+
}
|
|
371
|
+
return hash >>> 0;
|
|
372
|
+
}
|
|
373
|
+
|
|
334
374
|
function takeSnapshot(context: AgentContext, options: BuildOptions): StablePrefixSnapshot {
|
|
335
375
|
const systemPrompt = [...context.systemPrompt];
|
|
336
376
|
const tools = normalizeTools(context.tools, options.intentTracing) ?? [];
|
|
@@ -347,8 +387,42 @@ function normalizeImportedTools(tools: readonly Tool[], options: BuildOptions):
|
|
|
347
387
|
return cloneJson(normalizedTools);
|
|
348
388
|
}
|
|
349
389
|
|
|
350
|
-
function cloneJson<T>(value: T): T {
|
|
351
|
-
return
|
|
390
|
+
export function cloneJson<T>(value: T): T {
|
|
391
|
+
return cloneJsonValue(value) as T;
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
function cloneJsonValue(value: unknown, key = "", applyToJson = true): unknown {
|
|
395
|
+
if (value === null) return null;
|
|
396
|
+
const type = typeof value;
|
|
397
|
+
if (type === "number") return Number.isFinite(value) ? value : null;
|
|
398
|
+
// JSON.stringify drops function/symbol/undefined values (object props
|
|
399
|
+
// omitted, array elements become null via the array walk below).
|
|
400
|
+
if (type === "undefined" || type === "function" || type === "symbol") return undefined;
|
|
401
|
+
if (type !== "object") return value;
|
|
402
|
+
if (applyToJson) {
|
|
403
|
+
// JSON.stringify performs a single Get of `toJSON` per holder/key and
|
|
404
|
+
// serializes the returned replacement WITHOUT re-dispatching the
|
|
405
|
+
// replacement's own toJSON at the same level (nested properties still
|
|
406
|
+
// dispatch normally). Mirror that exactly to keep byte parity.
|
|
407
|
+
const toJSON = (value as { toJSON?: unknown }).toJSON;
|
|
408
|
+
if (typeof toJSON === "function") {
|
|
409
|
+
return cloneJsonValue(toJSON.call(value, key), key, false);
|
|
410
|
+
}
|
|
411
|
+
}
|
|
412
|
+
if (Array.isArray(value)) {
|
|
413
|
+
const cloned: unknown[] = new Array(value.length);
|
|
414
|
+
for (let i = 0; i < value.length; i++) {
|
|
415
|
+
const item = Object.hasOwn(value, i) ? cloneJsonValue(value[i], String(i)) : undefined;
|
|
416
|
+
cloned[i] = item === undefined ? null : item;
|
|
417
|
+
}
|
|
418
|
+
return cloned;
|
|
419
|
+
}
|
|
420
|
+
const cloned: Record<string, unknown> = {};
|
|
421
|
+
for (const key of Object.keys(value as object)) {
|
|
422
|
+
const clonedValue = cloneJsonValue((value as Record<string, unknown>)[key], key);
|
|
423
|
+
if (clonedValue !== undefined) cloned[key] = clonedValue;
|
|
424
|
+
}
|
|
425
|
+
return cloned;
|
|
352
426
|
}
|
|
353
427
|
|
|
354
428
|
function computeFingerprint(systemPrompt: string[], tools: Tool[], options: BuildOptions): string {
|
|
@@ -13,7 +13,6 @@ import {
|
|
|
13
13
|
type Model,
|
|
14
14
|
type Usage,
|
|
15
15
|
} from "@gajae-code/ai";
|
|
16
|
-
import { countTokens } from "@gajae-code/natives";
|
|
17
16
|
import { logger, prompt } from "@gajae-code/utils";
|
|
18
17
|
import { type AgentTelemetry, instrumentedCompleteSimple } from "../telemetry";
|
|
19
18
|
import type { AgentMessage, AgentTool } from "../types";
|
|
@@ -268,18 +267,97 @@ export function resolveThresholdTokens(
|
|
|
268
267
|
const IMAGE_TOKEN_ESTIMATE = 1200;
|
|
269
268
|
|
|
270
269
|
/**
|
|
271
|
-
*
|
|
272
|
-
*
|
|
273
|
-
*
|
|
270
|
+
* Lazily-required native `countTokens`. `@gajae-code/natives` dlopens a ~39MB
|
|
271
|
+
* addon; importing it at module scope would put that cost on every cold path
|
|
272
|
+
* that touches compaction exports (status line, print mode, context report).
|
|
273
|
+
* Deferring the require to the first context-changing call keeps the trivial
|
|
274
|
+
* `-p` / display paths native-free.
|
|
274
275
|
*/
|
|
275
|
-
|
|
276
|
+
let cachedNativeCountTokens: ((input: string | string[], encoding?: unknown) => number) | null = null;
|
|
277
|
+
|
|
278
|
+
function nativeCountTokens(fragments: string[]): number {
|
|
279
|
+
if (!cachedNativeCountTokens) {
|
|
280
|
+
const { createRequire } = require("node:module") as typeof import("node:module");
|
|
281
|
+
const requireFromHere = createRequire(import.meta.url);
|
|
282
|
+
const natives = requireFromHere("@gajae-code/natives") as {
|
|
283
|
+
countTokens: (input: string | string[], encoding?: unknown) => number;
|
|
284
|
+
};
|
|
285
|
+
cachedNativeCountTokens = natives.countTokens;
|
|
286
|
+
}
|
|
287
|
+
return cachedNativeCountTokens(fragments);
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
function countCollectedMessageFragments(collected: { fragments: string[]; extra: number }): number {
|
|
291
|
+
return nativeCountTokens(collected.fragments) + collected.extra;
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
/**
|
|
295
|
+
* Estimate token count for a message using the native o200k tokenizer.
|
|
296
|
+
* Exact for o200k only; an approximation for Anthropic/other model families
|
|
297
|
+
* (Anthropic doesn't publish a tokenizer) within ~5–10% on English/code text.
|
|
298
|
+
*
|
|
299
|
+
* This materializes the native BPE table (~50MB RSS) on first call. Use it
|
|
300
|
+
* only for context-changing decisions (compaction trigger/cut points, pruning
|
|
301
|
+
* budgets, branch summarization, fork-context seeding, context-limit
|
|
302
|
+
* enforcement). For display-only totals use
|
|
303
|
+
* {@link estimateMessageTokensHeuristic}.
|
|
304
|
+
*/
|
|
305
|
+
export function countMessageTokensNativeO200k(message: AgentMessage): number {
|
|
306
|
+
return countCollectedMessageFragments(collectMessageFragments(message));
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
/**
|
|
310
|
+
* Backwards-compatible alias for {@link countMessageTokensNativeO200k}.
|
|
311
|
+
* Existing callers treat this as the canonical message-token estimator for
|
|
312
|
+
* context-changing decisions.
|
|
313
|
+
*/
|
|
314
|
+
export const estimateTokens = countMessageTokensNativeO200k;
|
|
315
|
+
|
|
316
|
+
/**
|
|
317
|
+
* Average bytes per token for the cheap heuristic. ~4 bytes/token is the
|
|
318
|
+
* conventional approximation for English/code text under modern BPE
|
|
319
|
+
* vocabularies; it intentionally errs slightly low-precision in exchange for
|
|
320
|
+
* never touching the native tokenizer (and its ~50MB BPE table).
|
|
321
|
+
*/
|
|
322
|
+
const HEURISTIC_BYTES_PER_TOKEN = 4;
|
|
323
|
+
|
|
324
|
+
/**
|
|
325
|
+
* Cheap, native-free token estimate for a message. Suitable ONLY for
|
|
326
|
+
* display/init surfaces (status line, /context report, HUD totals) — never
|
|
327
|
+
* for context-changing decisions, which must use
|
|
328
|
+
* {@link countMessageTokensNativeO200k}.
|
|
329
|
+
*/
|
|
330
|
+
export function estimateMessageTokensHeuristic(message: AgentMessage): number {
|
|
331
|
+
const { fragments, extra } = collectMessageFragments(message);
|
|
332
|
+
let bytes = 0;
|
|
333
|
+
for (const fragment of fragments) {
|
|
334
|
+
bytes += fragment.length;
|
|
335
|
+
}
|
|
336
|
+
return extra + Math.ceil(bytes / HEURISTIC_BYTES_PER_TOKEN);
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
/**
|
|
340
|
+
* Cheap, native-free token estimate for plain string fragments. Display-only
|
|
341
|
+
* counterpart of the native `countTokens(fragments)` aggregate.
|
|
342
|
+
*/
|
|
343
|
+
export function estimateTextTokensHeuristic(fragments: string | readonly string[]): number {
|
|
344
|
+
if (typeof fragments === "string") return Math.ceil(fragments.length / HEURISTIC_BYTES_PER_TOKEN);
|
|
345
|
+
let bytes = 0;
|
|
346
|
+
for (const fragment of fragments) {
|
|
347
|
+
bytes += fragment.length;
|
|
348
|
+
}
|
|
349
|
+
return Math.ceil(bytes / HEURISTIC_BYTES_PER_TOKEN);
|
|
350
|
+
}
|
|
351
|
+
|
|
352
|
+
/** Shared content walk for both the native and heuristic estimators. */
|
|
353
|
+
function collectMessageFragments(message: AgentMessage): { fragments: string[]; extra: number } {
|
|
276
354
|
const fragments: string[] = [];
|
|
277
355
|
let extra = 0;
|
|
278
356
|
if ((message as { role?: string }).role === "bashExecution") {
|
|
279
357
|
const bash = message as { command?: unknown; output?: unknown };
|
|
280
358
|
if (typeof bash.command === "string") fragments.push(bash.command);
|
|
281
359
|
if (typeof bash.output === "string") fragments.push(bash.output);
|
|
282
|
-
return fragments
|
|
360
|
+
return { fragments, extra };
|
|
283
361
|
}
|
|
284
362
|
|
|
285
363
|
switch (message.role) {
|
|
@@ -331,20 +409,45 @@ export function estimateTokens(message: AgentMessage): number {
|
|
|
331
409
|
break;
|
|
332
410
|
}
|
|
333
411
|
default:
|
|
334
|
-
|
|
412
|
+
break;
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
return { fragments, extra };
|
|
416
|
+
}
|
|
417
|
+
|
|
418
|
+
function entryTokenFingerprint(
|
|
419
|
+
entry: SessionEntry,
|
|
420
|
+
message: AgentMessage,
|
|
421
|
+
collected: { fragments: string[]; extra: number },
|
|
422
|
+
): string {
|
|
423
|
+
const maybePruned = message as { prunedAt?: unknown };
|
|
424
|
+
let fingerprint = `${entry.type.length}:${entry.type}${(entry.id ?? "").length}:${entry.id ?? ""}${message.role.length}:${message.role}${String(collected.extra).length}:${String(collected.extra)}${collected.fragments.length}:`;
|
|
425
|
+
for (const fragment of collected.fragments) fingerprint += `${fragment.length}:${fragment}`;
|
|
426
|
+
if (maybePruned.prunedAt !== undefined) {
|
|
427
|
+
const prunedAt = String(maybePruned.prunedAt);
|
|
428
|
+
fingerprint += `prunedAt${prunedAt.length}:${prunedAt}`;
|
|
335
429
|
}
|
|
430
|
+
return fingerprint;
|
|
431
|
+
}
|
|
336
432
|
|
|
337
|
-
|
|
338
|
-
|
|
433
|
+
const entryTokenCache = new WeakMap<SessionEntry, { fingerprint: string; tokens: number }>();
|
|
434
|
+
|
|
435
|
+
export function estimateEntryTokens(entry: SessionEntry): number {
|
|
436
|
+
const msg = getMessageFromEntry(entry);
|
|
437
|
+
if (!msg) return 0;
|
|
438
|
+
const collected = collectMessageFragments(msg);
|
|
439
|
+
const fingerprint = entryTokenFingerprint(entry, msg, collected);
|
|
440
|
+
const cached = entryTokenCache.get(entry);
|
|
441
|
+
if (cached?.fingerprint === fingerprint) return cached.tokens;
|
|
442
|
+
const tokens = countCollectedMessageFragments(collected);
|
|
443
|
+
entryTokenCache.set(entry, { fingerprint, tokens });
|
|
444
|
+
return tokens;
|
|
339
445
|
}
|
|
340
446
|
|
|
341
|
-
function estimateEntriesTokens(entries: SessionEntry[], startIndex: number, endIndex: number): number {
|
|
447
|
+
export function estimateEntriesTokens(entries: SessionEntry[], startIndex: number, endIndex: number): number {
|
|
342
448
|
let total = 0;
|
|
343
449
|
for (let i = startIndex; i < endIndex; i++) {
|
|
344
|
-
|
|
345
|
-
if (msg) {
|
|
346
|
-
total += estimateTokens(msg);
|
|
347
|
-
}
|
|
450
|
+
total += estimateEntryTokens(entries[i]);
|
|
348
451
|
}
|
|
349
452
|
return total;
|
|
350
453
|
}
|
|
@@ -461,18 +564,23 @@ export function findCutPoint(
|
|
|
461
564
|
if (entry.type !== "message") continue;
|
|
462
565
|
|
|
463
566
|
// Estimate this message's size
|
|
464
|
-
const messageTokens =
|
|
567
|
+
const messageTokens = estimateEntryTokens(entry);
|
|
465
568
|
accumulatedTokens += messageTokens;
|
|
466
569
|
|
|
467
570
|
// Check if we've exceeded the budget
|
|
468
571
|
if (accumulatedTokens >= keepRecentTokens) {
|
|
469
572
|
// Find the closest valid cut point at or after this entry
|
|
573
|
+
let foundCutPoint = false;
|
|
470
574
|
for (let c = 0; c < cutPoints.length; c++) {
|
|
471
575
|
if (cutPoints[c] >= i) {
|
|
472
576
|
cutIndex = cutPoints[c];
|
|
577
|
+
foundCutPoint = true;
|
|
473
578
|
break;
|
|
474
579
|
}
|
|
475
580
|
}
|
|
581
|
+
if (!foundCutPoint) {
|
|
582
|
+
cutIndex = cutPoints[cutPoints.length - 1];
|
|
583
|
+
}
|
|
476
584
|
break;
|
|
477
585
|
}
|
|
478
586
|
}
|
|
@@ -24,6 +24,12 @@ export interface ModelChangeEntry extends SessionEntryBase {
|
|
|
24
24
|
model: string;
|
|
25
25
|
/** Role: "default", "smol", "slow", etc. Undefined treated as "default" */
|
|
26
26
|
role?: string;
|
|
27
|
+
/** Requested model before a runtime substitution/fallback, in "provider/modelId" format. */
|
|
28
|
+
previousModel?: string;
|
|
29
|
+
/** Machine-readable reason for runtime model substitution/fallback. */
|
|
30
|
+
reason?: string;
|
|
31
|
+
/** Effective thinking level when the change was recorded. */
|
|
32
|
+
thinkingLevel?: string | null;
|
|
27
33
|
}
|
|
28
34
|
|
|
29
35
|
export interface ServiceTierChangeEntry extends SessionEntryBase {
|
package/src/compaction/openai.ts
CHANGED
|
@@ -154,7 +154,7 @@ export function withOpenAiRemoteCompactionPreserveData(
|
|
|
154
154
|
// Input/output filtering for OpenAI compact endpoint
|
|
155
155
|
// ============================================================================
|
|
156
156
|
|
|
157
|
-
function estimateOpenAiCompactInputTokens(input: Array<Record<string, unknown>>, instructions: string): number {
|
|
157
|
+
export function estimateOpenAiCompactInputTokens(input: Array<Record<string, unknown>>, instructions: string): number {
|
|
158
158
|
let chars = instructions.length;
|
|
159
159
|
for (const item of input) {
|
|
160
160
|
chars += JSON.stringify(item).length;
|
|
@@ -200,22 +200,31 @@ function shouldKeepOpenAiCompactOutputItem(item: Record<string, unknown>): boole
|
|
|
200
200
|
return shouldKeepOpenAiCompactOutputUserMessage(item);
|
|
201
201
|
}
|
|
202
202
|
|
|
203
|
-
function trimOpenAiCompactInput(
|
|
203
|
+
export function trimOpenAiCompactInput(
|
|
204
204
|
input: Array<Record<string, unknown>>,
|
|
205
205
|
contextWindow: number,
|
|
206
206
|
instructions: string,
|
|
207
207
|
): Array<Record<string, unknown>> {
|
|
208
|
+
const itemLengths = input.map(item => JSON.stringify(item).length);
|
|
209
|
+
let chars = instructions.length;
|
|
210
|
+
for (const length of itemLengths) chars += length;
|
|
211
|
+
|
|
212
|
+
function removeAt(index: number): void {
|
|
213
|
+
chars -= itemLengths[index] ?? 0;
|
|
214
|
+
trimmed.splice(index, 1);
|
|
215
|
+
itemLengths.splice(index, 1);
|
|
216
|
+
}
|
|
208
217
|
const trimmed = [...input];
|
|
209
|
-
while (trimmed.length > 0 &&
|
|
218
|
+
while (trimmed.length > 0 && Math.ceil(chars / 4) > contextWindow) {
|
|
210
219
|
const last = trimmed[trimmed.length - 1];
|
|
211
220
|
if (last?.type === "function_call_output" || last?.type === "custom_tool_call_output") {
|
|
212
221
|
const callId = typeof last.call_id === "string" ? last.call_id : undefined;
|
|
213
222
|
const callType = last.type === "custom_tool_call_output" ? "custom_tool_call" : "function_call";
|
|
214
|
-
trimmed.
|
|
223
|
+
removeAt(trimmed.length - 1);
|
|
215
224
|
if (callId) {
|
|
216
225
|
const matchingCallIndex = trimmed.findLastIndex(item => item.type === callType && item.call_id === callId);
|
|
217
226
|
if (matchingCallIndex >= 0) {
|
|
218
|
-
|
|
227
|
+
removeAt(matchingCallIndex);
|
|
219
228
|
}
|
|
220
229
|
}
|
|
221
230
|
continue;
|
|
@@ -223,7 +232,7 @@ function trimOpenAiCompactInput(
|
|
|
223
232
|
if (!last || !shouldTrimOpenAiCompactInputItem(last)) {
|
|
224
233
|
break;
|
|
225
234
|
}
|
|
226
|
-
trimmed.
|
|
235
|
+
removeAt(trimmed.length - 1);
|
|
227
236
|
}
|
|
228
237
|
return trimmed;
|
|
229
238
|
}
|
|
@@ -1,10 +1,16 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Tool output pruning utilities for compaction.
|
|
3
|
+
*
|
|
4
|
+
* Candidate selection is staleness-aware: tool results that have been
|
|
5
|
+
* superseded by a later result for the same target (same file read again,
|
|
6
|
+
* same search re-run) or invalidated by a later successful edit/write to a
|
|
7
|
+
* covered file are pruned in preference to merely-old results. Protect-window
|
|
8
|
+
* and minimum-savings hysteresis semantics are unchanged.
|
|
3
9
|
*/
|
|
4
10
|
|
|
5
|
-
import type { ToolResultMessage } from "@gajae-code/ai";
|
|
11
|
+
import type { ToolCall, ToolResultMessage } from "@gajae-code/ai";
|
|
6
12
|
import type { AgentMessage } from "../types";
|
|
7
|
-
import {
|
|
13
|
+
import { estimateEntryTokens } from "./compaction";
|
|
8
14
|
import type { SessionEntry, SessionMessageEntry } from "./entries";
|
|
9
15
|
|
|
10
16
|
export interface PruneConfig {
|
|
@@ -14,23 +20,100 @@ export interface PruneConfig {
|
|
|
14
20
|
minimumSavings: number;
|
|
15
21
|
/** Tool names that should never be pruned. */
|
|
16
22
|
protectedTools: string[];
|
|
23
|
+
/**
|
|
24
|
+
* Tools in `protectedTools` whose protection is waived once the result is
|
|
25
|
+
* superseded (a later result for the same target, or a later successful
|
|
26
|
+
* edit/write to the covered file). The most recent result per target is
|
|
27
|
+
* never considered superseded. Optional; defaults to none.
|
|
28
|
+
*/
|
|
29
|
+
staleOverridableTools?: string[];
|
|
17
30
|
}
|
|
18
31
|
|
|
19
32
|
export const DEFAULT_PRUNE_CONFIG: PruneConfig = {
|
|
20
33
|
protectTokens: 40_000,
|
|
21
34
|
minimumSavings: 20_000,
|
|
22
35
|
protectedTools: ["skill", "read"],
|
|
36
|
+
staleOverridableTools: ["read"],
|
|
23
37
|
};
|
|
24
38
|
|
|
25
39
|
export interface PruneResult {
|
|
26
40
|
prunedCount: number;
|
|
27
41
|
tokensSaved: number;
|
|
42
|
+
/**
|
|
43
|
+
* The mutated message entries. Callers whose entry source returns
|
|
44
|
+
* materialized copies (not live references) must write these back into
|
|
45
|
+
* their canonical store by id.
|
|
46
|
+
*/
|
|
47
|
+
prunedEntries: SessionMessageEntry[];
|
|
28
48
|
}
|
|
29
49
|
|
|
30
|
-
|
|
50
|
+
const DIGEST_NOTICE_TOKEN_CAP_MULTIPLIER = 1.25;
|
|
51
|
+
|
|
52
|
+
function createGenericPrunedNotice(tokens: number): string {
|
|
31
53
|
return `[Output truncated - ${tokens} tokens]`;
|
|
32
54
|
}
|
|
33
55
|
|
|
56
|
+
function firstTextContent(message: ToolResultMessage): string {
|
|
57
|
+
if (typeof message.content === "string") return message.content;
|
|
58
|
+
const block = message.content.find(part => part.type === "text");
|
|
59
|
+
return block?.type === "text" ? block.text : "";
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
function firstErrorLine(text: string): string | undefined {
|
|
63
|
+
return text
|
|
64
|
+
.split(/\r?\n/)
|
|
65
|
+
.find(line => /error|failed|exception|panic/i.test(line))
|
|
66
|
+
?.trim();
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
function truncateField(value: string, maxLength: number): string {
|
|
70
|
+
if (value.length <= maxLength) return value;
|
|
71
|
+
if (maxLength <= 1) return "…";
|
|
72
|
+
return `${value.slice(0, maxLength - 1)}…`;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
function resultDigest(message: ToolResultMessage): string | undefined {
|
|
76
|
+
const toolName = message.toolName.toLowerCase();
|
|
77
|
+
const text = firstTextContent(message);
|
|
78
|
+
if (toolName === "bash") {
|
|
79
|
+
const details = message as { details?: { exitCode?: unknown } };
|
|
80
|
+
const exitCode =
|
|
81
|
+
typeof details.details?.exitCode === "number" ? details.details.exitCode : message.isError ? 1 : 0;
|
|
82
|
+
const tail = text.trim().split(/\r?\n/).filter(Boolean).at(-1) ?? "";
|
|
83
|
+
const error = firstErrorLine(text);
|
|
84
|
+
return [`exit=${exitCode}`, tail ? `tail=${tail}` : undefined, error ? `error=${error}` : undefined]
|
|
85
|
+
.filter((part): part is string => part !== undefined)
|
|
86
|
+
.join("; ");
|
|
87
|
+
}
|
|
88
|
+
if (toolName === "search" || toolName === "grep") {
|
|
89
|
+
const match = text.match(/(\d+)\s+matches?/i) ?? text.match(/totalMatches["']?:\s*(\d+)/i);
|
|
90
|
+
const files = text.match(/(\d+)\s+files?/i) ?? text.match(/filesWithMatches["']?:\s*(\d+)/i);
|
|
91
|
+
const error = firstErrorLine(text);
|
|
92
|
+
return (
|
|
93
|
+
[
|
|
94
|
+
match ? `matches=${match[1]}` : undefined,
|
|
95
|
+
files ? `files=${files[1]}` : undefined,
|
|
96
|
+
error ? `error=${error}` : undefined,
|
|
97
|
+
]
|
|
98
|
+
.filter((part): part is string => part !== undefined)
|
|
99
|
+
.join("; ") || "search digest unavailable"
|
|
100
|
+
);
|
|
101
|
+
}
|
|
102
|
+
return undefined;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
function createPrunedNotice(tokens: number, message?: ToolResultMessage): string {
|
|
106
|
+
const generic = createGenericPrunedNotice(tokens);
|
|
107
|
+
const digest = message ? resultDigest(message) : undefined;
|
|
108
|
+
if (!digest) return generic;
|
|
109
|
+
const genericTokens = Math.ceil(generic.length / 4);
|
|
110
|
+
const maxTokens = Math.max(genericTokens, Math.floor(genericTokens * DIGEST_NOTICE_TOKEN_CAP_MULTIPLIER));
|
|
111
|
+
const prefix = `[Output truncated - ${tokens} tokens; `;
|
|
112
|
+
const suffix = "]";
|
|
113
|
+
const maxChars = Math.max(0, maxTokens * 4 - prefix.length - suffix.length);
|
|
114
|
+
return `${prefix}${truncateField(digest, maxChars)}${suffix}`;
|
|
115
|
+
}
|
|
116
|
+
|
|
34
117
|
function getToolResultMessage(entry: SessionEntry): ToolResultMessage | undefined {
|
|
35
118
|
if (entry.type !== "message") return undefined;
|
|
36
119
|
const message = entry.message as AgentMessage;
|
|
@@ -38,55 +121,311 @@ function getToolResultMessage(entry: SessionEntry): ToolResultMessage | undefine
|
|
|
38
121
|
return message as ToolResultMessage;
|
|
39
122
|
}
|
|
40
123
|
|
|
41
|
-
function estimatePrunedSavings(tokens: number): number {
|
|
42
|
-
const noticeTokens = Math.ceil(
|
|
124
|
+
function estimatePrunedSavings(tokens: number, notice: string): number {
|
|
125
|
+
const noticeTokens = Math.ceil(notice.length / 4);
|
|
43
126
|
return Math.max(0, tokens - noticeTokens);
|
|
44
127
|
}
|
|
45
128
|
|
|
129
|
+
const EDIT_TOOL_NAMES = new Set(["edit", "write", "apply_patch", "ast_edit"]);
|
|
130
|
+
|
|
131
|
+
/** Extract the file-path argument from a tool call, when the tool has one. */
|
|
132
|
+
function toolCallPath(call: ToolCall): string | undefined {
|
|
133
|
+
const args = call.arguments;
|
|
134
|
+
const path = args.path ?? args.file_path ?? args.filePath;
|
|
135
|
+
return typeof path === "string" && path.length > 0 ? path : undefined;
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
/**
|
|
139
|
+
* `*** Add|Update|Delete File: <path>` headers open a hunk; `*** Move to:
|
|
140
|
+
* <path>` attaches a rename destination to the current hunk. Move
|
|
141
|
+
* destinations count as touched paths: a rename onto a file invalidates
|
|
142
|
+
* earlier reads of that destination.
|
|
143
|
+
*/
|
|
144
|
+
const APPLY_PATCH_HEADER = /^\*\*\* (?:((?:Add|Update|Delete) File)|(Move to)): (.+)$/gm;
|
|
145
|
+
|
|
146
|
+
/**
|
|
147
|
+
* Paths touched by an edit-class tool call, grouped per hunk so a failed
|
|
148
|
+
* hunk can be excluded wholesale (its rename destination included). Most
|
|
149
|
+
* edit tools carry a single path argument; apply_patch envelopes carry an
|
|
150
|
+
* `input` string with per-file headers instead. The envelope shape can
|
|
151
|
+
* arrive under the custom `apply_patch` tool OR the regular `edit` tool
|
|
152
|
+
* (providers without custom-tool support fall back to the JSON function), so
|
|
153
|
+
* any edit-class call with a string `input` is parsed for headers.
|
|
154
|
+
*/
|
|
155
|
+
function editToolPathGroups(call: ToolCall): string[][] {
|
|
156
|
+
const path = toolCallPath(call);
|
|
157
|
+
if (path !== undefined) return [[path]];
|
|
158
|
+
const input = call.arguments.input;
|
|
159
|
+
if (typeof input !== "string") return [];
|
|
160
|
+
const groups: string[][] = [];
|
|
161
|
+
for (const match of input.matchAll(APPLY_PATCH_HEADER)) {
|
|
162
|
+
const headerPath = match[3]?.trim();
|
|
163
|
+
if (!headerPath) continue;
|
|
164
|
+
const isMoveTo = match[2] !== undefined;
|
|
165
|
+
if (isMoveTo && groups.length > 0) {
|
|
166
|
+
groups[groups.length - 1].push(headerPath);
|
|
167
|
+
} else {
|
|
168
|
+
groups.push([headerPath]);
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
return groups;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/**
|
|
175
|
+
* Trailing read selectors (`:50`, `:50-200`, `:50+150`, `:5-16,960-973`,
|
|
176
|
+
* `:raw`, `:conflicts`), possibly stacked (`:2-4:raw`). Stripped to resolve
|
|
177
|
+
* the underlying file for edit invalidation.
|
|
178
|
+
*/
|
|
179
|
+
const READ_SELECTOR_SUFFIX = /:(?:raw|conflicts|\d+(?:[-+]\d+)?(?:,\d+(?:[-+]\d+)?)*)$/;
|
|
180
|
+
|
|
181
|
+
/** Base file path of a read target with any line/mode selectors stripped. */
|
|
182
|
+
function readBasePath(path: string): string {
|
|
183
|
+
let base = path;
|
|
184
|
+
while (READ_SELECTOR_SUFFIX.test(base)) {
|
|
185
|
+
base = base.replace(READ_SELECTOR_SUFFIX, "");
|
|
186
|
+
}
|
|
187
|
+
return base;
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
/**
|
|
191
|
+
* Stable identity for "the same logical lookup": same tool re-targeting the
|
|
192
|
+
* same subject. A later result with the same key supersedes earlier ones.
|
|
193
|
+
* Keys are canonical JSON tuples so user-controlled text (patterns, paths)
|
|
194
|
+
* can never collide via delimiter ambiguity. Search keys include pagination
|
|
195
|
+
* (`skip`) and result-shaping flags (`i`, `gitignore`): a later page or a
|
|
196
|
+
* differently-shaped search complements earlier output, it does not replace it.
|
|
197
|
+
*/
|
|
198
|
+
function toolTargetKey(call: ToolCall): string | undefined {
|
|
199
|
+
const path = toolCallPath(call);
|
|
200
|
+
if (path !== undefined) return JSON.stringify([call.name, "path", path]);
|
|
201
|
+
const pattern = call.arguments.pattern;
|
|
202
|
+
if (typeof pattern === "string" && pattern.length > 0) {
|
|
203
|
+
const paths = call.arguments.paths;
|
|
204
|
+
const pathList = Array.isArray(paths) ? paths.filter((p): p is string => typeof p === "string") : [];
|
|
205
|
+
const skip = typeof call.arguments.skip === "number" ? call.arguments.skip : 0;
|
|
206
|
+
const caseInsensitive = call.arguments.i === true;
|
|
207
|
+
const gitignore = call.arguments.gitignore !== false;
|
|
208
|
+
return JSON.stringify([call.name, "pattern", pattern, pathList, skip, caseInsensitive, gitignore]);
|
|
209
|
+
}
|
|
210
|
+
return undefined;
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
/**
|
|
214
|
+
* Files actually mutated according to a tool result's details. Used for
|
|
215
|
+
* AST-edit-shaped results (`ast_edit` direct-apply and the hidden `resolve`
|
|
216
|
+
* apply step), which report `{ applied: true, files: [...] }` — the resolve
|
|
217
|
+
* tool nests that payload under `details.sourceResultDetails`. Conservative:
|
|
218
|
+
* returns nothing unless the details explicitly mark the change as applied.
|
|
219
|
+
* Checked even on `isError` results: a stale-preview apply reports an error
|
|
220
|
+
* while still having mutated the listed files.
|
|
221
|
+
*/
|
|
222
|
+
function resultDetailFiles(message: ToolResultMessage): string[] {
|
|
223
|
+
const raw = message.details as { applied?: unknown; files?: unknown; sourceResultDetails?: unknown } | undefined;
|
|
224
|
+
const candidates = [raw, raw?.sourceResultDetails as { applied?: unknown; files?: unknown } | undefined];
|
|
225
|
+
for (const details of candidates) {
|
|
226
|
+
if (details?.applied === true && Array.isArray(details.files)) {
|
|
227
|
+
return details.files.filter((file): file is string => typeof file === "string" && file.length > 0);
|
|
228
|
+
}
|
|
229
|
+
}
|
|
230
|
+
return [];
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
/**
|
|
234
|
+
* Paths that FAILED in a per-file edit result (`details.perFileResults`) and
|
|
235
|
+
* were NOT mutated by any same-path entry. Multi-file apply_patch catches
|
|
236
|
+
* per-file failures and still returns a non-error result; a purely-failed
|
|
237
|
+
* path was not mutated and must not stale reads. But apply_patch can emit
|
|
238
|
+
* multiple entries for the same path (e.g. several hunks): if any same-path
|
|
239
|
+
* entry succeeded the file still mutated, so it must NOT be suppressed.
|
|
240
|
+
* Conservative: only an entry explicitly marked `isError === true` counts as
|
|
241
|
+
* a failure; anything else (including ambiguous/malformed entries) counts as
|
|
242
|
+
* a success and keeps the path out of the suppression set.
|
|
243
|
+
*/
|
|
244
|
+
function failedEditPaths(message: ToolResultMessage): Set<string> {
|
|
245
|
+
const details = message.details as { perFileResults?: unknown } | undefined;
|
|
246
|
+
const perFile = details?.perFileResults;
|
|
247
|
+
if (!Array.isArray(perFile)) return new Set();
|
|
248
|
+
const failed = new Set<string>();
|
|
249
|
+
const succeeded = new Set<string>();
|
|
250
|
+
for (const item of perFile) {
|
|
251
|
+
const entry = item as { path?: unknown; isError?: unknown };
|
|
252
|
+
if (typeof entry?.path !== "string") continue;
|
|
253
|
+
if (entry.isError === true) failed.add(entry.path);
|
|
254
|
+
else succeeded.add(entry.path);
|
|
255
|
+
}
|
|
256
|
+
// A path mutated if any same-path entry succeeded, even when another
|
|
257
|
+
// same-path entry failed; drop those from the suppression set.
|
|
258
|
+
for (const path of succeeded) failed.delete(path);
|
|
259
|
+
return failed;
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
/**
|
|
263
|
+
* Concrete file path a `read` result actually came from, when the tool
|
|
264
|
+
* reported one (`details.resolvedPath`). Suffix resolution can map a bare
|
|
265
|
+
* filename argument onto a different concrete path.
|
|
266
|
+
*/
|
|
267
|
+
function readResolvedPath(message: ToolResultMessage): string | undefined {
|
|
268
|
+
const details = message.details as { resolvedPath?: unknown } | undefined;
|
|
269
|
+
const resolved = details?.resolvedPath;
|
|
270
|
+
return typeof resolved === "string" && resolved.length > 0 ? resolved : undefined;
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
interface StalenessIndex {
|
|
274
|
+
/** Entry indices of toolResults superseded by a later same-target result or a later edit. */
|
|
275
|
+
staleResultIndices: Set<number>;
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
/**
|
|
279
|
+
* Build a staleness index over session entries (oldest -> newest):
|
|
280
|
+
* - a toolResult is stale when a later non-error toolResult shares its target key;
|
|
281
|
+
* - a `read` result is stale when a later non-error edit/write touches its file.
|
|
282
|
+
* The most recent result per target is never stale.
|
|
283
|
+
*/
|
|
284
|
+
function buildStalenessIndex(entries: SessionEntry[]): StalenessIndex {
|
|
285
|
+
const callsById = new Map<string, ToolCall>();
|
|
286
|
+
for (const entry of entries) {
|
|
287
|
+
if (entry.type !== "message") continue;
|
|
288
|
+
const message = entry.message as AgentMessage;
|
|
289
|
+
if (message.role !== "assistant") continue;
|
|
290
|
+
for (const content of message.content) {
|
|
291
|
+
if (content.type === "toolCall") callsById.set(content.id, content);
|
|
292
|
+
}
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
const lastResultIndexByKey = new Map<string, number>();
|
|
296
|
+
const resultMeta = new Map<number, { key?: string; call: ToolCall; message: ToolResultMessage }>();
|
|
297
|
+
const lastEditIndexByPath = new Map<string, number>();
|
|
298
|
+
|
|
299
|
+
for (let i = 0; i < entries.length; i++) {
|
|
300
|
+
const message = getToolResultMessage(entries[i]);
|
|
301
|
+
if (!message) continue;
|
|
302
|
+
const call = callsById.get(message.toolCallId);
|
|
303
|
+
if (!call) continue;
|
|
304
|
+
|
|
305
|
+
// AST edits mutate files when previews are applied via the hidden
|
|
306
|
+
// `resolve` tool; the call args carry globs, not concrete paths. Both
|
|
307
|
+
// tools report actually-touched files in result details. Collected
|
|
308
|
+
// BEFORE the error gate: a stale-preview apply reports an error while
|
|
309
|
+
// still having mutated the listed files.
|
|
310
|
+
if (call.name === "resolve" || call.name === "ast_edit") {
|
|
311
|
+
for (const editPath of resultDetailFiles(message)) {
|
|
312
|
+
lastEditIndexByPath.set(editPath, i);
|
|
313
|
+
}
|
|
314
|
+
}
|
|
315
|
+
if (message.isError) continue;
|
|
316
|
+
|
|
317
|
+
const key = toolTargetKey(call);
|
|
318
|
+
resultMeta.set(i, { key, call, message });
|
|
319
|
+
if (key !== undefined) lastResultIndexByKey.set(key, i);
|
|
320
|
+
if (EDIT_TOOL_NAMES.has(call.name)) {
|
|
321
|
+
// Per-file edit results record failures in details.perFileResults;
|
|
322
|
+
// a failed hunk mutated nothing, so exclude its whole path group
|
|
323
|
+
// (rename destination included) from touched paths.
|
|
324
|
+
const failed = failedEditPaths(message);
|
|
325
|
+
for (const group of editToolPathGroups(call)) {
|
|
326
|
+
if (group.some(groupPath => failed.has(groupPath))) continue;
|
|
327
|
+
for (const editPath of group) {
|
|
328
|
+
lastEditIndexByPath.set(editPath, i);
|
|
329
|
+
}
|
|
330
|
+
}
|
|
331
|
+
}
|
|
332
|
+
}
|
|
333
|
+
|
|
334
|
+
const staleResultIndices = new Set<number>();
|
|
335
|
+
for (const [index, meta] of resultMeta) {
|
|
336
|
+
if (meta.key !== undefined) {
|
|
337
|
+
const lastIndex = lastResultIndexByKey.get(meta.key);
|
|
338
|
+
if (lastIndex !== undefined && lastIndex > index) {
|
|
339
|
+
staleResultIndices.add(index);
|
|
340
|
+
continue;
|
|
341
|
+
}
|
|
342
|
+
}
|
|
343
|
+
if (meta.call.name === "read") {
|
|
344
|
+
// Check both the call argument (selectors stripped) and the resolved
|
|
345
|
+
// path from result details: suffix resolution can map a bare filename
|
|
346
|
+
// onto a different concrete path, and edits may use either form.
|
|
347
|
+
const lookupPaths = new Set<string>();
|
|
348
|
+
const argPath = toolCallPath(meta.call);
|
|
349
|
+
if (argPath !== undefined) lookupPaths.add(readBasePath(argPath));
|
|
350
|
+
const resolved = readResolvedPath(meta.message);
|
|
351
|
+
if (resolved !== undefined) lookupPaths.add(resolved);
|
|
352
|
+
for (const lookupPath of lookupPaths) {
|
|
353
|
+
const editIndex = lastEditIndexByPath.get(lookupPath);
|
|
354
|
+
if (editIndex !== undefined && editIndex > index) {
|
|
355
|
+
staleResultIndices.add(index);
|
|
356
|
+
break;
|
|
357
|
+
}
|
|
358
|
+
}
|
|
359
|
+
}
|
|
360
|
+
}
|
|
361
|
+
|
|
362
|
+
return { staleResultIndices };
|
|
363
|
+
}
|
|
364
|
+
|
|
46
365
|
export function pruneToolOutputs(entries: SessionEntry[], config: PruneConfig = DEFAULT_PRUNE_CONFIG): PruneResult {
|
|
47
366
|
let accumulatedTokens = 0;
|
|
48
367
|
let tokensSaved = 0;
|
|
49
368
|
let prunedCount = 0;
|
|
50
369
|
|
|
51
|
-
const
|
|
370
|
+
const { staleResultIndices } = buildStalenessIndex(entries);
|
|
371
|
+
const staleOverridable = new Set(config.staleOverridableTools ?? []);
|
|
372
|
+
const candidates: Array<{ entry: SessionMessageEntry; tokens: number; notice: string; savings: number }> = [];
|
|
52
373
|
|
|
53
374
|
for (let i = entries.length - 1; i >= 0; i--) {
|
|
54
375
|
const entry = entries[i];
|
|
55
376
|
const message = getToolResultMessage(entry);
|
|
56
377
|
if (!message) continue;
|
|
57
378
|
|
|
58
|
-
const tokens =
|
|
59
|
-
const
|
|
379
|
+
const tokens = estimateEntryTokens(entry);
|
|
380
|
+
const isStale = staleResultIndices.has(i);
|
|
381
|
+
// Staleness waives protected-tool immunity for overridable tools
|
|
382
|
+
// (e.g. a superseded `read`); the most recent result per target is
|
|
383
|
+
// never stale, so the latest read of each file stays protected.
|
|
384
|
+
const isProtected =
|
|
385
|
+
config.protectedTools.includes(message.toolName) && !(isStale && staleOverridable.has(message.toolName));
|
|
60
386
|
|
|
61
387
|
if (message.prunedAt !== undefined) {
|
|
62
388
|
accumulatedTokens += tokens;
|
|
63
389
|
continue;
|
|
64
390
|
}
|
|
65
391
|
|
|
66
|
-
|
|
392
|
+
// Stale results are prunable even inside the recency protect window —
|
|
393
|
+
// they are superseded, so recency no longer implies relevance. They
|
|
394
|
+
// still count toward window accounting so non-stale protection is
|
|
395
|
+
// unchanged.
|
|
396
|
+
const insideProtectWindow = accumulatedTokens < config.protectTokens;
|
|
397
|
+
if ((insideProtectWindow && !isStale) || isProtected) {
|
|
67
398
|
accumulatedTokens += tokens;
|
|
68
399
|
continue;
|
|
69
400
|
}
|
|
70
401
|
|
|
71
|
-
|
|
402
|
+
const notice = createPrunedNotice(tokens, message);
|
|
403
|
+
candidates.push({
|
|
404
|
+
entry: entry as SessionMessageEntry,
|
|
405
|
+
tokens,
|
|
406
|
+
notice,
|
|
407
|
+
savings: estimatePrunedSavings(tokens, notice),
|
|
408
|
+
});
|
|
72
409
|
accumulatedTokens += tokens;
|
|
73
410
|
}
|
|
74
411
|
|
|
75
412
|
for (const candidate of candidates) {
|
|
76
|
-
tokensSaved +=
|
|
413
|
+
tokensSaved += candidate.savings;
|
|
77
414
|
}
|
|
78
415
|
|
|
79
416
|
if (tokensSaved < config.minimumSavings || candidates.length === 0) {
|
|
80
|
-
return { prunedCount: 0, tokensSaved: 0 };
|
|
417
|
+
return { prunedCount: 0, tokensSaved: 0, prunedEntries: [] };
|
|
81
418
|
}
|
|
82
419
|
|
|
83
420
|
const prunedAt = Date.now();
|
|
421
|
+
const prunedEntries: SessionMessageEntry[] = [];
|
|
84
422
|
for (const candidate of candidates) {
|
|
85
423
|
const message = candidate.entry.message as ToolResultMessage;
|
|
86
|
-
message.content = [{ type: "text", text:
|
|
424
|
+
message.content = [{ type: "text", text: candidate.notice }];
|
|
87
425
|
message.prunedAt = prunedAt;
|
|
426
|
+
prunedEntries.push(candidate.entry);
|
|
88
427
|
prunedCount++;
|
|
89
428
|
}
|
|
90
429
|
|
|
91
|
-
return { prunedCount, tokensSaved };
|
|
430
|
+
return { prunedCount, tokensSaved, prunedEntries };
|
|
92
431
|
}
|
package/src/types.ts
CHANGED
|
@@ -189,6 +189,9 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
|
|
|
189
189
|
*/
|
|
190
190
|
onAssistantMessageEvent?: (message: AssistantMessage, event: AssistantMessageEvent) => void;
|
|
191
191
|
|
|
192
|
+
/** Called for non-content tool-choice incapability stream events. */
|
|
193
|
+
onToolChoiceIncapability?: (event: Extract<AssistantMessageEvent, { type: "toolChoiceIncapability" }>) => void;
|
|
194
|
+
|
|
192
195
|
/**
|
|
193
196
|
* Called when GPT-5 Harmony protocol leakage is detected and mitigated.
|
|
194
197
|
*/
|