@gajae-code/agent-core 0.4.2 → 0.4.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +6 -0
- package/dist/types/agent.d.ts +6 -0
- package/dist/types/compaction/compaction.d.ts +13 -4
- package/package.json +4 -4
- package/src/agent.ts +15 -0
- package/src/compaction/compaction.ts +29 -7
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,12 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [0.4.3] - 2026-06-10
|
|
6
|
+
|
|
7
|
+
### Fixed
|
|
8
|
+
|
|
9
|
+
- Separated a model's total context window from its safe input/prompt-packing budget in the compaction threshold. `effectiveReserveTokens`, `resolveThresholdTokens`, and `shouldCompact` now accept an optional `maxOutputTokens` and reserve at least that completion budget, so a large-output model (e.g. 400K context / 128K max output) caps input near 272K instead of 340K and cannot overflow the total window with reserved output ([#442](https://github.com/Yeachan-Heo/gajae-code/issues/442)).
|
|
10
|
+
|
|
5
11
|
## [0.4.0] - 2026-06-06
|
|
6
12
|
|
|
7
13
|
### Changed
|
package/dist/types/agent.d.ts
CHANGED
|
@@ -346,6 +346,12 @@ export declare class Agent {
|
|
|
346
346
|
* Used by dequeue keybinding.
|
|
347
347
|
*/
|
|
348
348
|
popLastFollowUp(): AgentMessage | undefined;
|
|
349
|
+
/** Remove queued steering+follow-up messages matching `predicate`, preserving order of the rest. */
|
|
350
|
+
removeQueuedMessages(predicate: (message: AgentMessage) => boolean): {
|
|
351
|
+
steering: number;
|
|
352
|
+
followUp: number;
|
|
353
|
+
total: number;
|
|
354
|
+
};
|
|
349
355
|
clearMessages(): void;
|
|
350
356
|
abort(): void;
|
|
351
357
|
/**
|
|
@@ -50,14 +50,23 @@ export declare function calculatePromptTokens(usage: Usage): number;
|
|
|
50
50
|
*/
|
|
51
51
|
export declare function getLastAssistantUsage(entries: SessionEntry[]): Usage | undefined;
|
|
52
52
|
/**
|
|
53
|
-
* Effective reserve:
|
|
53
|
+
* Effective reserve: the largest of 15% of the context window, the configured floor,
|
|
54
|
+
* and the model's reserved completion budget (`maxOutputTokens`).
|
|
55
|
+
*
|
|
56
|
+
* Reserving `maxOutputTokens` keeps the safe input/prompt-packing budget below the
|
|
57
|
+
* *total* context window for models whose completion reservation exceeds the 15%
|
|
58
|
+
* floor (e.g. a 400K-context model with 128K max output reserves 128K, not 60K, so
|
|
59
|
+
* input is capped near 272K instead of 340K).
|
|
54
60
|
*/
|
|
55
|
-
export declare function effectiveReserveTokens(contextWindow: number, settings: CompactionSettings): number;
|
|
61
|
+
export declare function effectiveReserveTokens(contextWindow: number, settings: CompactionSettings, maxOutputTokens?: number): number;
|
|
56
62
|
/**
|
|
57
63
|
* Check if compaction should trigger based on context usage.
|
|
64
|
+
*
|
|
65
|
+
* `maxOutputTokens` is the model's reserved completion budget; it is excluded from
|
|
66
|
+
* the safe input budget so prompt + reserved output cannot exceed the total window.
|
|
58
67
|
*/
|
|
59
|
-
export declare function shouldCompact(contextTokens: number, contextWindow: number, settings: CompactionSettings): boolean;
|
|
60
|
-
export declare function resolveThresholdTokens(contextWindow: number, settings: CompactionSettings): number;
|
|
68
|
+
export declare function shouldCompact(contextTokens: number, contextWindow: number, settings: CompactionSettings, maxOutputTokens?: number): boolean;
|
|
69
|
+
export declare function resolveThresholdTokens(contextWindow: number, settings: CompactionSettings, maxOutputTokens?: number): number;
|
|
61
70
|
/**
|
|
62
71
|
* Estimate token count for a message using cl100k_base via the native
|
|
63
72
|
* tokenizer. This is not Anthropic's first-party tokenizer (Anthropic doesn't
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@gajae-code/agent-core",
|
|
4
|
-
"version": "0.4.
|
|
4
|
+
"version": "0.4.3",
|
|
5
5
|
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
|
|
6
6
|
"homepage": "https://gaebal-gajae.dev",
|
|
7
7
|
"author": "Yeachan-Heo",
|
|
@@ -35,9 +35,9 @@
|
|
|
35
35
|
"fmt": "biome format --write ."
|
|
36
36
|
},
|
|
37
37
|
"dependencies": {
|
|
38
|
-
"@gajae-code/ai": "0.4.
|
|
39
|
-
"@gajae-code/natives": "0.4.
|
|
40
|
-
"@gajae-code/utils": "0.4.
|
|
38
|
+
"@gajae-code/ai": "0.4.3",
|
|
39
|
+
"@gajae-code/natives": "0.4.3",
|
|
40
|
+
"@gajae-code/utils": "0.4.3",
|
|
41
41
|
"@opentelemetry/api": "^1.9.0"
|
|
42
42
|
},
|
|
43
43
|
"devDependencies": {
|
package/src/agent.ts
CHANGED
|
@@ -943,6 +943,21 @@ export class Agent {
|
|
|
943
943
|
return this.#followUpQueue.pop();
|
|
944
944
|
}
|
|
945
945
|
|
|
946
|
+
/** Remove queued steering+follow-up messages matching `predicate`, preserving order of the rest. */
|
|
947
|
+
removeQueuedMessages(predicate: (message: AgentMessage) => boolean): {
|
|
948
|
+
steering: number;
|
|
949
|
+
followUp: number;
|
|
950
|
+
total: number;
|
|
951
|
+
} {
|
|
952
|
+
const beforeSteering = this.#steeringQueue.length;
|
|
953
|
+
const beforeFollowUp = this.#followUpQueue.length;
|
|
954
|
+
this.#steeringQueue = this.#steeringQueue.filter(m => !predicate(m));
|
|
955
|
+
this.#followUpQueue = this.#followUpQueue.filter(m => !predicate(m));
|
|
956
|
+
const steering = beforeSteering - this.#steeringQueue.length;
|
|
957
|
+
const followUp = beforeFollowUp - this.#followUpQueue.length;
|
|
958
|
+
return { steering, followUp, total: steering + followUp };
|
|
959
|
+
}
|
|
960
|
+
|
|
946
961
|
clearMessages() {
|
|
947
962
|
this.#state.messages = [];
|
|
948
963
|
}
|
|
@@ -203,22 +203,44 @@ export function getLastAssistantUsage(entries: SessionEntry[]): Usage | undefine
|
|
|
203
203
|
}
|
|
204
204
|
|
|
205
205
|
/**
|
|
206
|
-
* Effective reserve:
|
|
206
|
+
* Effective reserve: the largest of 15% of the context window, the configured floor,
|
|
207
|
+
* and the model's reserved completion budget (`maxOutputTokens`).
|
|
208
|
+
*
|
|
209
|
+
* Reserving `maxOutputTokens` keeps the safe input/prompt-packing budget below the
|
|
210
|
+
* *total* context window for models whose completion reservation exceeds the 15%
|
|
211
|
+
* floor (e.g. a 400K-context model with 128K max output reserves 128K, not 60K, so
|
|
212
|
+
* input is capped near 272K instead of 340K).
|
|
207
213
|
*/
|
|
208
|
-
export function effectiveReserveTokens(
|
|
209
|
-
|
|
214
|
+
export function effectiveReserveTokens(
|
|
215
|
+
contextWindow: number,
|
|
216
|
+
settings: CompactionSettings,
|
|
217
|
+
maxOutputTokens = 0,
|
|
218
|
+
): number {
|
|
219
|
+
return Math.max(Math.floor(contextWindow * 0.15), settings.reserveTokens, Math.max(0, maxOutputTokens));
|
|
210
220
|
}
|
|
211
221
|
|
|
212
222
|
/**
|
|
213
223
|
* Check if compaction should trigger based on context usage.
|
|
224
|
+
*
|
|
225
|
+
* `maxOutputTokens` is the model's reserved completion budget; it is excluded from
|
|
226
|
+
* the safe input budget so prompt + reserved output cannot exceed the total window.
|
|
214
227
|
*/
|
|
215
|
-
export function shouldCompact(
|
|
228
|
+
export function shouldCompact(
|
|
229
|
+
contextTokens: number,
|
|
230
|
+
contextWindow: number,
|
|
231
|
+
settings: CompactionSettings,
|
|
232
|
+
maxOutputTokens = 0,
|
|
233
|
+
): boolean {
|
|
216
234
|
if (!settings.enabled || settings.strategy === "off" || contextWindow <= 0) return false;
|
|
217
|
-
const thresholdTokens = resolveThresholdTokens(contextWindow, settings);
|
|
235
|
+
const thresholdTokens = resolveThresholdTokens(contextWindow, settings, maxOutputTokens);
|
|
218
236
|
return contextTokens > thresholdTokens;
|
|
219
237
|
}
|
|
220
238
|
|
|
221
|
-
export function resolveThresholdTokens(
|
|
239
|
+
export function resolveThresholdTokens(
|
|
240
|
+
contextWindow: number,
|
|
241
|
+
settings: CompactionSettings,
|
|
242
|
+
maxOutputTokens = 0,
|
|
243
|
+
): number {
|
|
222
244
|
// Fixed token limit takes priority over percentage
|
|
223
245
|
const thresholdTokens = settings.thresholdTokens;
|
|
224
246
|
if (typeof thresholdTokens === "number" && Number.isFinite(thresholdTokens) && thresholdTokens > 0) {
|
|
@@ -229,7 +251,7 @@ export function resolveThresholdTokens(contextWindow: number, settings: Compacti
|
|
|
229
251
|
// Percentage-based threshold
|
|
230
252
|
const thresholdPercent = settings.thresholdPercent;
|
|
231
253
|
if (typeof thresholdPercent !== "number" || !Number.isFinite(thresholdPercent) || thresholdPercent <= 0) {
|
|
232
|
-
return contextWindow - effectiveReserveTokens(contextWindow, settings);
|
|
254
|
+
return contextWindow - effectiveReserveTokens(contextWindow, settings, maxOutputTokens);
|
|
233
255
|
}
|
|
234
256
|
const clampedThresholdPercent = Math.min(99, Math.max(1, thresholdPercent));
|
|
235
257
|
return Math.floor(contextWindow * (clampedThresholdPercent / 100));
|