@gajae-code/agent-core 0.5.2 → 0.5.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/dist/types/agent.d.ts +6 -0
- package/dist/types/compaction/branch-summarization.d.ts +10 -1
- package/dist/types/compaction/compaction.d.ts +51 -1
- package/package.json +4 -4
- package/src/agent.ts +13 -1
- package/src/append-only-context.ts +53 -39
- package/src/compaction/branch-summarization.ts +22 -3
- package/src/compaction/compaction.ts +127 -5
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,18 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [0.5.4] - 2026-06-17
|
|
6
|
+
|
|
7
|
+
### Fixed
|
|
8
|
+
|
|
9
|
+
- Maintenance one-shot LLM calls now preserve active provider session state and the configured WebSocket transport preference. `SummaryOptions`, `HandoffOptions`, and `GenerateBranchSummaryOptions` accept `sessionId`, `providerSessionState`, and `preferWebsockets`, and `generateSummary`, `generateShortSummary`, `generateTurnPrefixSummary`, `generateHandoff`, `generateBranchSummary`, and `compact()` forward them through to `completeSimple` — previously these fields were dropped, so Codex/OpenAI-compatible compaction summaries, handoff generation, and branch summaries fell back to HTTP/SSE and lost `session_id` affinity even with `providers.openaiWebsockets: "on"`. Split-turn compaction now runs its history and turn-prefix summaries sequentially when they share a single provider WebSocket session, avoiding `websocket request already in progress`; non-WebSocket sessions still run them in parallel. `Agent` exposes a `preferWebsockets` getter so callers can forward the live transport preference (#736).
|
|
10
|
+
|
|
11
|
+
## [0.5.3] - 2026-06-16
|
|
12
|
+
|
|
13
|
+
### Fixed
|
|
14
|
+
|
|
15
|
+
- Bounded agent context growth, compaction, and token accounting for long-running sessions: `appendMessage` pushes in place instead of rebuilding the array; the append-only context keeps rolling per-message hashes instead of rescanning the full digest; an emergency-compaction floor that cannot be disabled now surfaces its reason; `getSessionStats` is single-pass; and `nativeCountTokens` skips the synchronous ~39 MB BPE tokenizer above a 2 MiB input cap, falling back to the cheap heuristic (#717).
|
|
16
|
+
|
|
5
17
|
## [0.5.2] - 2026-06-15
|
|
6
18
|
|
|
7
19
|
### Fixed
|
package/dist/types/agent.d.ts
CHANGED
|
@@ -193,6 +193,12 @@ export declare class Agent {
|
|
|
193
193
|
set sessionId(value: string | undefined);
|
|
194
194
|
get providerSessionId(): string | undefined;
|
|
195
195
|
set providerSessionId(value: string | undefined);
|
|
196
|
+
/**
|
|
197
|
+
* Whether websocket transport is preferred when the provider implementation
|
|
198
|
+
* supports it. Read by maintenance one-shot calls (compaction, handoff,
|
|
199
|
+
* branch summary) so they forward the same transport preference as live turns.
|
|
200
|
+
*/
|
|
201
|
+
get preferWebsockets(): boolean | undefined;
|
|
196
202
|
/**
|
|
197
203
|
* Static metadata forwarded to every API request when no resolver is installed
|
|
198
204
|
* (e.g. `metadata.user_id` for Anthropic session attribution). Setting this
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
* When navigating to a different point in the session tree, this generates
|
|
5
5
|
* a summary of the branch being left so context isn't lost.
|
|
6
6
|
*/
|
|
7
|
-
import type { Model } from "@gajae-code/ai";
|
|
7
|
+
import type { Model, ProviderSessionState } from "@gajae-code/ai";
|
|
8
8
|
import { type AgentTelemetry } from "../telemetry";
|
|
9
9
|
import type { AgentMessage } from "../types";
|
|
10
10
|
import type { ReadonlySessionManager, SessionEntry } from "./entries";
|
|
@@ -57,6 +57,15 @@ export interface GenerateBranchSummaryOptions {
|
|
|
57
57
|
* wrapped in an OTEL chat span tagged with `pi.gen_ai.oneshot.kind = "branch_summary"`.
|
|
58
58
|
*/
|
|
59
59
|
telemetry?: AgentTelemetry;
|
|
60
|
+
/**
|
|
61
|
+
* Provider session affinity id forwarded to the branch summary LLM call so it
|
|
62
|
+
* reuses the live turn's provider/WebSocket session.
|
|
63
|
+
*/
|
|
64
|
+
sessionId?: string;
|
|
65
|
+
/** Shared provider state map so the branch summary call reuses session-scoped transport/session caches. */
|
|
66
|
+
providerSessionState?: Map<string, ProviderSessionState>;
|
|
67
|
+
/** Hint that websocket transport should be preferred when supported by the provider implementation. */
|
|
68
|
+
preferWebsockets?: boolean;
|
|
60
69
|
}
|
|
61
70
|
/**
|
|
62
71
|
* Collect entries that should be summarized when navigating from one position to another.
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
* Pure functions for compaction logic. The session manager handles I/O,
|
|
5
5
|
* and after compaction the session is reloaded.
|
|
6
6
|
*/
|
|
7
|
-
import { type MessageAttribution, type Model, type Usage } from "@gajae-code/ai";
|
|
7
|
+
import { type MessageAttribution, type Model, type ProviderSessionState, type Usage } from "@gajae-code/ai";
|
|
8
8
|
import { type AgentTelemetry } from "../telemetry";
|
|
9
9
|
import type { AgentMessage, AgentTool } from "../types";
|
|
10
10
|
import type { SessionEntry } from "./entries";
|
|
@@ -66,6 +66,37 @@ export declare function effectiveReserveTokens(contextWindow: number, settings:
|
|
|
66
66
|
* the safe input budget so prompt + reserved output cannot exceed the total window.
|
|
67
67
|
*/
|
|
68
68
|
export declare function shouldCompact(contextTokens: number, contextWindow: number, settings: CompactionSettings, maxOutputTokens?: number): boolean;
|
|
69
|
+
/** Reason a compaction was triggered. `token` is the normal user-configurable path; the rest are emergency floors. */
|
|
70
|
+
export type CompactionTriggerReason = "token" | "heap" | "providerBytes" | "messageCount" | "imageBytes";
|
|
71
|
+
/** A point-in-time resource sample. Supplied by an injectable sampler so tests never read real RSS. */
|
|
72
|
+
export interface EmergencyCompactionSample {
|
|
73
|
+
/** Resident heap bytes (e.g. process.memoryUsage().heapUsed). */
|
|
74
|
+
heapUsedBytes: number;
|
|
75
|
+
/** Approximate serialized provider-context bytes. */
|
|
76
|
+
providerBytes: number;
|
|
77
|
+
/** Provider-visible message count. */
|
|
78
|
+
messageCount: number;
|
|
79
|
+
/** Approximate inline image bytes in the provider context. */
|
|
80
|
+
imageBytes: number;
|
|
81
|
+
}
|
|
82
|
+
export interface EmergencyCompactionLimits {
|
|
83
|
+
heapUsedBytes: number;
|
|
84
|
+
providerBytes: number;
|
|
85
|
+
messageCount: number;
|
|
86
|
+
imageBytes: number;
|
|
87
|
+
}
|
|
88
|
+
/**
|
|
89
|
+
* Non-disableable emergency floors. These sit well above normal usage and exist so a
|
|
90
|
+
* long session on weak hardware compacts before OOM even when token-based compaction is
|
|
91
|
+
* disabled or its threshold is set too high. They are NOT user-tunable down to zero.
|
|
92
|
+
*/
|
|
93
|
+
export declare const DEFAULT_EMERGENCY_COMPACTION_LIMITS: EmergencyCompactionLimits;
|
|
94
|
+
/**
|
|
95
|
+
* Returns the first emergency limit exceeded (heap > providerBytes > imageBytes > messageCount),
|
|
96
|
+
* or null when none is. Pure and sampler-injected; the caller routes the result through the
|
|
97
|
+
* normal pair-safe `compact()` cut logic so a tool_use/tool_result pair is never split.
|
|
98
|
+
*/
|
|
99
|
+
export declare function emergencyCompactionReason(sample: EmergencyCompactionSample, limits?: EmergencyCompactionLimits): CompactionTriggerReason | null;
|
|
69
100
|
export declare function resolveThresholdTokens(contextWindow: number, settings: CompactionSettings, maxOutputTokens?: number): number;
|
|
70
101
|
/**
|
|
71
102
|
* Estimate token count for a message using the native o200k tokenizer.
|
|
@@ -151,6 +182,16 @@ export interface SummaryOptions {
|
|
|
151
182
|
*/
|
|
152
183
|
telemetry?: AgentTelemetry;
|
|
153
184
|
authCredentialType?: "api_key" | "oauth";
|
|
185
|
+
/**
|
|
186
|
+
* Provider session affinity id forwarded to the maintenance LLM call so it
|
|
187
|
+
* reuses the live turn's provider/WebSocket session (matches the
|
|
188
|
+
* `providerSessionId ?? sessionId` the agent loop sends for normal turns).
|
|
189
|
+
*/
|
|
190
|
+
sessionId?: string;
|
|
191
|
+
/** Shared provider state map so maintenance calls reuse session-scoped transport/session caches. */
|
|
192
|
+
providerSessionState?: Map<string, ProviderSessionState>;
|
|
193
|
+
/** Hint that websocket transport should be preferred when supported by the provider implementation. */
|
|
194
|
+
preferWebsockets?: boolean;
|
|
154
195
|
}
|
|
155
196
|
export declare function generateSummary(currentMessages: AgentMessage[], model: Model, reserveTokens: number, apiKey: string, signal?: AbortSignal, customInstructions?: string, previousSummary?: string, options?: SummaryOptions): Promise<string>;
|
|
156
197
|
export interface HandoffOptions {
|
|
@@ -168,6 +209,15 @@ export interface HandoffOptions {
|
|
|
168
209
|
*/
|
|
169
210
|
telemetry?: AgentTelemetry;
|
|
170
211
|
authCredentialType?: "api_key" | "oauth";
|
|
212
|
+
/**
|
|
213
|
+
* Provider session affinity id forwarded to the handoff LLM call so it
|
|
214
|
+
* reuses the live turn's provider/WebSocket session.
|
|
215
|
+
*/
|
|
216
|
+
sessionId?: string;
|
|
217
|
+
/** Shared provider state map so the handoff call reuses session-scoped transport/session caches. */
|
|
218
|
+
providerSessionState?: Map<string, ProviderSessionState>;
|
|
219
|
+
/** Hint that websocket transport should be preferred when supported by the provider implementation. */
|
|
220
|
+
preferWebsockets?: boolean;
|
|
171
221
|
}
|
|
172
222
|
export declare function renderHandoffPrompt(customInstructions?: string): string;
|
|
173
223
|
export declare function generateHandoff(messages: AgentMessage[], model: Model, apiKey: string, options: HandoffOptions, signal?: AbortSignal): Promise<string>;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@gajae-code/agent-core",
|
|
4
|
-
"version": "0.5.
|
|
4
|
+
"version": "0.5.4",
|
|
5
5
|
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
|
|
6
6
|
"homepage": "https://gaebal-gajae.dev",
|
|
7
7
|
"author": "Yeachan-Heo",
|
|
@@ -35,9 +35,9 @@
|
|
|
35
35
|
"fmt": "biome format --write ."
|
|
36
36
|
},
|
|
37
37
|
"dependencies": {
|
|
38
|
-
"@gajae-code/ai": "0.5.
|
|
39
|
-
"@gajae-code/natives": "0.5.
|
|
40
|
-
"@gajae-code/utils": "0.5.
|
|
38
|
+
"@gajae-code/ai": "0.5.4",
|
|
39
|
+
"@gajae-code/natives": "0.5.4",
|
|
40
|
+
"@gajae-code/utils": "0.5.4",
|
|
41
41
|
"@opentelemetry/api": "^1.9.0"
|
|
42
42
|
},
|
|
43
43
|
"devDependencies": {
|
package/src/agent.ts
CHANGED
|
@@ -408,6 +408,15 @@ export class Agent {
|
|
|
408
408
|
this.#providerSessionId = value;
|
|
409
409
|
}
|
|
410
410
|
|
|
411
|
+
/**
|
|
412
|
+
* Whether websocket transport is preferred when the provider implementation
|
|
413
|
+
* supports it. Read by maintenance one-shot calls (compaction, handoff,
|
|
414
|
+
* branch summary) so they forward the same transport preference as live turns.
|
|
415
|
+
*/
|
|
416
|
+
get preferWebsockets(): boolean | undefined {
|
|
417
|
+
return this.#preferWebsockets;
|
|
418
|
+
}
|
|
419
|
+
|
|
411
420
|
/**
|
|
412
421
|
* Static metadata forwarded to every API request when no resolver is installed
|
|
413
422
|
* (e.g. `metadata.user_id` for Anthropic session attribution). Setting this
|
|
@@ -827,7 +836,10 @@ export class Agent {
|
|
|
827
836
|
}
|
|
828
837
|
|
|
829
838
|
appendMessage(m: AgentMessage) {
|
|
830
|
-
|
|
839
|
+
// In-place push (not [...messages, m]): appending M messages over a session of
|
|
840
|
+
// N is O(N+M), not O(M*N). Consumers read state.messages fresh; run() snapshots
|
|
841
|
+
// via slice() at the API boundary, so no caller relies on per-append array identity.
|
|
842
|
+
this.#state.messages.push(m);
|
|
831
843
|
}
|
|
832
844
|
|
|
833
845
|
popMessage(): AgentMessage | undefined {
|
|
@@ -199,8 +199,8 @@ export class AppendOnlyContextManager {
|
|
|
199
199
|
readonly log = new AppendOnlyLog();
|
|
200
200
|
/** How many normalized messages were synced into the log as of the last sync. */
|
|
201
201
|
#lastSyncCount = 0;
|
|
202
|
-
/**
|
|
203
|
-
#
|
|
202
|
+
/** Per-synced-message content hashes (rolling digest). Detects in-place rewrites without retaining a full serialized-history string. */
|
|
203
|
+
#syncedHashes: (number | bigint)[] = [];
|
|
204
204
|
/** Number of provider-normalized messages that were seeded before child-local messages. */
|
|
205
205
|
#seededPrefixCount = 0;
|
|
206
206
|
|
|
@@ -236,35 +236,41 @@ export class AppendOnlyContextManager {
|
|
|
236
236
|
const includesSeedPrefix =
|
|
237
237
|
seededPrefixLength > 0 &&
|
|
238
238
|
normalizedMessages.length >= seededPrefixLength &&
|
|
239
|
-
this.#
|
|
240
|
-
this.#computeDigestRange(this.log.entries(), 0, seededPrefixLength).source;
|
|
239
|
+
this.#rangeHashesEqual(normalizedMessages, this.log.entries(), seededPrefixLength);
|
|
241
240
|
const messagesToSync =
|
|
242
241
|
seededPrefixLength > 0 && !includesSeedPrefix
|
|
243
242
|
? [...this.log.entries().slice(0, seededPrefixLength), ...normalizedMessages]
|
|
244
243
|
: normalizedMessages;
|
|
245
244
|
|
|
246
|
-
// Detect in-place rewrites of already-synced messages
|
|
245
|
+
// Detect in-place rewrites of already-synced messages via per-message content
|
|
246
|
+
// hashes (no retained full serialized-history string; F5).
|
|
247
247
|
if (
|
|
248
248
|
this.#lastSyncCount > 0 &&
|
|
249
249
|
this.#lastSyncCount <= messagesToSync.length &&
|
|
250
|
-
this.#
|
|
250
|
+
this.#prefixChanged(messagesToSync, this.#lastSyncCount)
|
|
251
251
|
) {
|
|
252
252
|
if (this.#seededPrefixCount > 0) {
|
|
253
|
-
|
|
253
|
+
// F9: a seeded fork whose inherited prefix changed (e.g. after compaction)
|
|
254
|
+
// rebases onto the new provider context instead of throwing.
|
|
255
|
+
this.#rebaseToBaseline(normalizedMessages);
|
|
256
|
+
return;
|
|
254
257
|
}
|
|
255
258
|
this.log.clear();
|
|
256
259
|
this.#lastSyncCount = 0;
|
|
260
|
+
this.#syncedHashes = [];
|
|
257
261
|
}
|
|
258
262
|
|
|
259
|
-
// Compaction — array shrunk. Seeded forks preserve the inherited prefix
|
|
260
|
-
//
|
|
261
|
-
//
|
|
263
|
+
// Compaction — array shrunk. Seeded forks preserve the inherited prefix and
|
|
264
|
+
// append child-local deltas, so a shorter child array is not a compaction signal
|
|
265
|
+
// while a seed prefix is active; a genuine seeded compaction rebases (F9).
|
|
262
266
|
if (messagesToSync.length < this.#lastSyncCount) {
|
|
263
267
|
if (this.#seededPrefixCount > 0) {
|
|
264
|
-
|
|
268
|
+
this.#rebaseToBaseline(normalizedMessages);
|
|
269
|
+
return;
|
|
265
270
|
}
|
|
266
271
|
this.log.clear();
|
|
267
272
|
this.#lastSyncCount = 0;
|
|
273
|
+
this.#syncedHashes = [];
|
|
268
274
|
}
|
|
269
275
|
|
|
270
276
|
const newMsgs = messagesToSync.slice(this.#lastSyncCount);
|
|
@@ -273,7 +279,7 @@ export class AppendOnlyContextManager {
|
|
|
273
279
|
}
|
|
274
280
|
|
|
275
281
|
this.#lastSyncCount = messagesToSync.length;
|
|
276
|
-
this.#
|
|
282
|
+
this.#syncedHashes = this.#hashRange(messagesToSync, 0, messagesToSync.length);
|
|
277
283
|
}
|
|
278
284
|
|
|
279
285
|
seedNormalizedMessages(messages: readonly Message[], options?: { reset?: boolean }): void {
|
|
@@ -284,7 +290,7 @@ export class AppendOnlyContextManager {
|
|
|
284
290
|
this.log.clear();
|
|
285
291
|
this.log.extend(clonedMessages);
|
|
286
292
|
this.#lastSyncCount = clonedMessages.length;
|
|
287
|
-
this.#
|
|
293
|
+
this.#syncedHashes = this.#hashRange(clonedMessages, 0, clonedMessages.length);
|
|
288
294
|
this.#seededPrefixCount = clonedMessages.length;
|
|
289
295
|
}
|
|
290
296
|
|
|
@@ -293,7 +299,7 @@ export class AppendOnlyContextManager {
|
|
|
293
299
|
this.prefix.invalidate();
|
|
294
300
|
this.log.clear();
|
|
295
301
|
this.#lastSyncCount = 0;
|
|
296
|
-
this.#
|
|
302
|
+
this.#syncedHashes = [];
|
|
297
303
|
this.#seededPrefixCount = 0;
|
|
298
304
|
}
|
|
299
305
|
|
|
@@ -301,7 +307,7 @@ export class AppendOnlyContextManager {
|
|
|
301
307
|
resetSyncCursor(): void {
|
|
302
308
|
this.log.clear();
|
|
303
309
|
this.#lastSyncCount = 0;
|
|
304
|
-
this.#
|
|
310
|
+
this.#syncedHashes = [];
|
|
305
311
|
this.#seededPrefixCount = 0;
|
|
306
312
|
}
|
|
307
313
|
|
|
@@ -321,28 +327,45 @@ export class AppendOnlyContextManager {
|
|
|
321
327
|
this.prefix.invalidate();
|
|
322
328
|
this.log.clear();
|
|
323
329
|
this.#lastSyncCount = 0;
|
|
324
|
-
this.#
|
|
330
|
+
this.#syncedHashes = [];
|
|
325
331
|
this.#seededPrefixCount = 0;
|
|
326
332
|
this.prefix.build(context, options);
|
|
327
333
|
}
|
|
328
334
|
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
335
|
+
#hashMessage(message: unknown): number | bigint {
|
|
336
|
+
return hashSource(JSON.stringify(message) ?? "null");
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
#hashRange(messages: readonly unknown[], start: number, end: number): (number | bigint)[] {
|
|
340
|
+
const out: (number | bigint)[] = [];
|
|
341
|
+
for (let i = start; i < end; i++) out.push(this.#hashMessage(messages[i]));
|
|
342
|
+
return out;
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
/** True when the first `count` messages of `a` and `b` are content-equal by per-message hash. */
|
|
346
|
+
#rangeHashesEqual(a: readonly unknown[], b: readonly unknown[], count: number): boolean {
|
|
347
|
+
for (let i = 0; i < count; i++) {
|
|
348
|
+
if (this.#hashMessage(a[i]) !== this.#hashMessage(b[i])) return false;
|
|
349
|
+
}
|
|
350
|
+
return true;
|
|
336
351
|
}
|
|
337
352
|
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
353
|
+
/** True when any of the first `count` already-synced messages changed content (in-place rewrite). */
|
|
354
|
+
#prefixChanged(messages: readonly unknown[], count: number): boolean {
|
|
355
|
+
if (count > this.#syncedHashes.length) return false;
|
|
356
|
+
for (let i = 0; i < count; i++) {
|
|
357
|
+
if (this.#hashMessage(messages[i]) !== this.#syncedHashes[i]) return true;
|
|
343
358
|
}
|
|
344
|
-
|
|
345
|
-
|
|
359
|
+
return false;
|
|
360
|
+
}
|
|
361
|
+
|
|
362
|
+
/** F9: reset the seeded log to a new provider-visible baseline (seeded compaction/rebase). */
|
|
363
|
+
#rebaseToBaseline(messages: readonly unknown[]): void {
|
|
364
|
+
this.log.clear();
|
|
365
|
+
this.log.extend([...messages]);
|
|
366
|
+
this.#lastSyncCount = messages.length;
|
|
367
|
+
this.#seededPrefixCount = 0;
|
|
368
|
+
this.#syncedHashes = this.#hashRange(messages, 0, messages.length);
|
|
346
369
|
}
|
|
347
370
|
}
|
|
348
371
|
|
|
@@ -350,15 +373,6 @@ export class AppendOnlyContextManager {
|
|
|
350
373
|
// Snapshot helpers
|
|
351
374
|
// ---------------------------------------------------------------------------
|
|
352
375
|
|
|
353
|
-
type MessageDigest = {
|
|
354
|
-
hash: number | bigint;
|
|
355
|
-
source: string;
|
|
356
|
-
};
|
|
357
|
-
|
|
358
|
-
function emptyMessageDigest(): MessageDigest {
|
|
359
|
-
return { hash: hashSource("[]"), source: "[]" };
|
|
360
|
-
}
|
|
361
|
-
|
|
362
376
|
function hashSource(source: string): number | bigint {
|
|
363
377
|
return typeof Bun !== "undefined" ? Bun.hash(source) : hashString32(source);
|
|
364
378
|
}
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
* a summary of the branch being left so context isn't lost.
|
|
6
6
|
*/
|
|
7
7
|
|
|
8
|
-
import type { Model } from "@gajae-code/ai";
|
|
8
|
+
import type { Model, ProviderSessionState } from "@gajae-code/ai";
|
|
9
9
|
import { prompt } from "@gajae-code/utils";
|
|
10
10
|
import { type AgentTelemetry, instrumentedCompleteSimple } from "../telemetry";
|
|
11
11
|
import type { AgentMessage } from "../types";
|
|
@@ -86,6 +86,15 @@ export interface GenerateBranchSummaryOptions {
|
|
|
86
86
|
* wrapped in an OTEL chat span tagged with `pi.gen_ai.oneshot.kind = "branch_summary"`.
|
|
87
87
|
*/
|
|
88
88
|
telemetry?: AgentTelemetry;
|
|
89
|
+
/**
|
|
90
|
+
* Provider session affinity id forwarded to the branch summary LLM call so it
|
|
91
|
+
* reuses the live turn's provider/WebSocket session.
|
|
92
|
+
*/
|
|
93
|
+
sessionId?: string;
|
|
94
|
+
/** Shared provider state map so the branch summary call reuses session-scoped transport/session caches. */
|
|
95
|
+
providerSessionState?: Map<string, ProviderSessionState>;
|
|
96
|
+
/** Hint that websocket transport should be preferred when supported by the provider implementation. */
|
|
97
|
+
preferWebsockets?: boolean;
|
|
89
98
|
}
|
|
90
99
|
|
|
91
100
|
// ============================================================================
|
|
@@ -274,7 +283,17 @@ export async function generateBranchSummary(
|
|
|
274
283
|
entries: SessionEntry[],
|
|
275
284
|
options: GenerateBranchSummaryOptions,
|
|
276
285
|
): Promise<BranchSummaryResult> {
|
|
277
|
-
const {
|
|
286
|
+
const {
|
|
287
|
+
model,
|
|
288
|
+
apiKey,
|
|
289
|
+
signal,
|
|
290
|
+
customInstructions,
|
|
291
|
+
reserveTokens = 16384,
|
|
292
|
+
metadata,
|
|
293
|
+
sessionId,
|
|
294
|
+
providerSessionState,
|
|
295
|
+
preferWebsockets,
|
|
296
|
+
} = options;
|
|
278
297
|
|
|
279
298
|
// Token budget = context window minus reserved space for prompt + response
|
|
280
299
|
const contextWindow = model.contextWindow || 128000;
|
|
@@ -307,7 +326,7 @@ export async function generateBranchSummary(
|
|
|
307
326
|
const response = await instrumentedCompleteSimple(
|
|
308
327
|
model,
|
|
309
328
|
{ systemPrompt: [SUMMARIZATION_SYSTEM_PROMPT], messages: summarizationMessages },
|
|
310
|
-
{ apiKey, signal, maxTokens: 2048, metadata },
|
|
329
|
+
{ apiKey, signal, maxTokens: 2048, metadata, sessionId, providerSessionState, preferWebsockets },
|
|
311
330
|
{ telemetry: options.telemetry, oneshotKind: "branch_summary" },
|
|
312
331
|
);
|
|
313
332
|
|
|
@@ -12,6 +12,7 @@ import {
|
|
|
12
12
|
type Message,
|
|
13
13
|
type MessageAttribution,
|
|
14
14
|
type Model,
|
|
15
|
+
type ProviderSessionState,
|
|
15
16
|
type Usage,
|
|
16
17
|
} from "@gajae-code/ai";
|
|
17
18
|
import { isCompiledBinary, logger, prompt } from "@gajae-code/utils";
|
|
@@ -236,6 +237,56 @@ export function shouldCompact(
|
|
|
236
237
|
return contextTokens > thresholdTokens;
|
|
237
238
|
}
|
|
238
239
|
|
|
240
|
+
/** Reason a compaction was triggered. `token` is the normal user-configurable path; the rest are emergency floors. */
|
|
241
|
+
export type CompactionTriggerReason = "token" | "heap" | "providerBytes" | "messageCount" | "imageBytes";
|
|
242
|
+
|
|
243
|
+
/** A point-in-time resource sample. Supplied by an injectable sampler so tests never read real RSS. */
|
|
244
|
+
export interface EmergencyCompactionSample {
|
|
245
|
+
/** Resident heap bytes (e.g. process.memoryUsage().heapUsed). */
|
|
246
|
+
heapUsedBytes: number;
|
|
247
|
+
/** Approximate serialized provider-context bytes. */
|
|
248
|
+
providerBytes: number;
|
|
249
|
+
/** Provider-visible message count. */
|
|
250
|
+
messageCount: number;
|
|
251
|
+
/** Approximate inline image bytes in the provider context. */
|
|
252
|
+
imageBytes: number;
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
export interface EmergencyCompactionLimits {
|
|
256
|
+
heapUsedBytes: number;
|
|
257
|
+
providerBytes: number;
|
|
258
|
+
messageCount: number;
|
|
259
|
+
imageBytes: number;
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
/**
|
|
263
|
+
* Non-disableable emergency floors. These sit well above normal usage and exist so a
|
|
264
|
+
* long session on weak hardware compacts before OOM even when token-based compaction is
|
|
265
|
+
* disabled or its threshold is set too high. They are NOT user-tunable down to zero.
|
|
266
|
+
*/
|
|
267
|
+
export const DEFAULT_EMERGENCY_COMPACTION_LIMITS: EmergencyCompactionLimits = {
|
|
268
|
+
heapUsedBytes: 1_536 * 1024 * 1024, // 1.5 GiB resident heap
|
|
269
|
+
providerBytes: 24 * 1024 * 1024, // 24 MiB serialized provider context
|
|
270
|
+
messageCount: 4000,
|
|
271
|
+
imageBytes: 64 * 1024 * 1024, // 64 MiB inline image bytes
|
|
272
|
+
};
|
|
273
|
+
|
|
274
|
+
/**
|
|
275
|
+
* Returns the first emergency limit exceeded (heap > providerBytes > imageBytes > messageCount),
|
|
276
|
+
* or null when none is. Pure and sampler-injected; the caller routes the result through the
|
|
277
|
+
* normal pair-safe `compact()` cut logic so a tool_use/tool_result pair is never split.
|
|
278
|
+
*/
|
|
279
|
+
export function emergencyCompactionReason(
|
|
280
|
+
sample: EmergencyCompactionSample,
|
|
281
|
+
limits: EmergencyCompactionLimits = DEFAULT_EMERGENCY_COMPACTION_LIMITS,
|
|
282
|
+
): CompactionTriggerReason | null {
|
|
283
|
+
if (sample.heapUsedBytes > limits.heapUsedBytes) return "heap";
|
|
284
|
+
if (sample.providerBytes > limits.providerBytes) return "providerBytes";
|
|
285
|
+
if (sample.imageBytes > limits.imageBytes) return "imageBytes";
|
|
286
|
+
if (sample.messageCount > limits.messageCount) return "messageCount";
|
|
287
|
+
return null;
|
|
288
|
+
}
|
|
289
|
+
|
|
239
290
|
export function resolveThresholdTokens(
|
|
240
291
|
contextWindow: number,
|
|
241
292
|
settings: CompactionSettings,
|
|
@@ -301,7 +352,18 @@ function nativeTokenizerEntrypoint(): string {
|
|
|
301
352
|
return isCompiledBinary() ? COMPILED_NATIVE_TOKENIZER_ENTRYPOINT : SOURCE_NATIVE_TOKENIZER_ENTRYPOINT;
|
|
302
353
|
}
|
|
303
354
|
|
|
355
|
+
/** Max total fragment chars sent to the synchronous native tokenizer (F22). */
|
|
356
|
+
const MAX_NATIVE_TOKENIZE_CHARS = 2 * 1024 * 1024;
|
|
357
|
+
|
|
304
358
|
function nativeCountTokens(fragments: string[]): number {
|
|
359
|
+
let totalChars = 0;
|
|
360
|
+
for (const fragment of fragments) totalChars += fragment.length;
|
|
361
|
+
if (totalChars > MAX_NATIVE_TOKENIZE_CHARS) {
|
|
362
|
+
// F22: skip the synchronous native BPE tokenizer (materializes a ~39MB table and is
|
|
363
|
+
// O(text)) on pathologically large inputs; the cheap chars/token heuristic is more
|
|
364
|
+
// than accurate enough for size/budget decisions and never blocks the event loop.
|
|
365
|
+
return estimateTextTokensHeuristic(fragments);
|
|
366
|
+
}
|
|
305
367
|
if (!cachedNativeCountTokens) {
|
|
306
368
|
const natives = requireFromCompaction(nativeTokenizerEntrypoint()) as NativeTokenizerModule;
|
|
307
369
|
cachedNativeCountTokens = natives.countTokens;
|
|
@@ -674,6 +736,16 @@ export interface SummaryOptions {
|
|
|
674
736
|
*/
|
|
675
737
|
telemetry?: AgentTelemetry;
|
|
676
738
|
authCredentialType?: "api_key" | "oauth";
|
|
739
|
+
/**
|
|
740
|
+
* Provider session affinity id forwarded to the maintenance LLM call so it
|
|
741
|
+
* reuses the live turn's provider/WebSocket session (matches the
|
|
742
|
+
* `providerSessionId ?? sessionId` the agent loop sends for normal turns).
|
|
743
|
+
*/
|
|
744
|
+
sessionId?: string;
|
|
745
|
+
/** Shared provider state map so maintenance calls reuse session-scoped transport/session caches. */
|
|
746
|
+
providerSessionState?: Map<string, ProviderSessionState>;
|
|
747
|
+
/** Hint that websocket transport should be preferred when supported by the provider implementation. */
|
|
748
|
+
preferWebsockets?: boolean;
|
|
677
749
|
}
|
|
678
750
|
|
|
679
751
|
export async function generateSummary(
|
|
@@ -740,6 +812,9 @@ export async function generateSummary(
|
|
|
740
812
|
reasoning: Effort.High,
|
|
741
813
|
initiatorOverride: options?.initiatorOverride,
|
|
742
814
|
metadata: options?.metadata,
|
|
815
|
+
sessionId: options?.sessionId,
|
|
816
|
+
providerSessionState: options?.providerSessionState,
|
|
817
|
+
preferWebsockets: options?.preferWebsockets,
|
|
743
818
|
},
|
|
744
819
|
{ telemetry: options?.telemetry, oneshotKind: "compaction_summary" },
|
|
745
820
|
);
|
|
@@ -775,6 +850,15 @@ export interface HandoffOptions {
|
|
|
775
850
|
*/
|
|
776
851
|
telemetry?: AgentTelemetry;
|
|
777
852
|
authCredentialType?: "api_key" | "oauth";
|
|
853
|
+
/**
|
|
854
|
+
* Provider session affinity id forwarded to the handoff LLM call so it
|
|
855
|
+
* reuses the live turn's provider/WebSocket session.
|
|
856
|
+
*/
|
|
857
|
+
sessionId?: string;
|
|
858
|
+
/** Shared provider state map so the handoff call reuses session-scoped transport/session caches. */
|
|
859
|
+
providerSessionState?: Map<string, ProviderSessionState>;
|
|
860
|
+
/** Hint that websocket transport should be preferred when supported by the provider implementation. */
|
|
861
|
+
preferWebsockets?: boolean;
|
|
778
862
|
}
|
|
779
863
|
|
|
780
864
|
export function renderHandoffPrompt(customInstructions?: string): string {
|
|
@@ -816,6 +900,9 @@ export async function generateHandoff(
|
|
|
816
900
|
toolChoice: "none",
|
|
817
901
|
initiatorOverride: options.initiatorOverride,
|
|
818
902
|
metadata: options.metadata,
|
|
903
|
+
sessionId: options.sessionId,
|
|
904
|
+
providerSessionState: options.providerSessionState,
|
|
905
|
+
preferWebsockets: options.preferWebsockets,
|
|
819
906
|
},
|
|
820
907
|
{ telemetry: options.telemetry, oneshotKind: "handoff" },
|
|
821
908
|
);
|
|
@@ -875,6 +962,9 @@ async function generateShortSummary(
|
|
|
875
962
|
reasoning: Effort.High,
|
|
876
963
|
initiatorOverride: options?.initiatorOverride,
|
|
877
964
|
metadata: options?.metadata,
|
|
965
|
+
sessionId: options?.sessionId,
|
|
966
|
+
providerSessionState: options?.providerSessionState,
|
|
967
|
+
preferWebsockets: options?.preferWebsockets,
|
|
878
968
|
},
|
|
879
969
|
{ telemetry: options?.telemetry, oneshotKind: "compaction_short_summary" },
|
|
880
970
|
);
|
|
@@ -1059,6 +1149,9 @@ export async function compact(
|
|
|
1059
1149
|
metadata: options?.metadata,
|
|
1060
1150
|
convertToLlm: options?.convertToLlm,
|
|
1061
1151
|
telemetry: options?.telemetry,
|
|
1152
|
+
sessionId: options?.sessionId,
|
|
1153
|
+
providerSessionState: options?.providerSessionState,
|
|
1154
|
+
preferWebsockets: options?.preferWebsockets,
|
|
1062
1155
|
};
|
|
1063
1156
|
|
|
1064
1157
|
let preserveData = withOpenAiRemoteCompactionPreserveData(previousPreserveData, undefined);
|
|
@@ -1098,9 +1191,22 @@ export async function compact(
|
|
|
1098
1191
|
// Generate summaries (can be parallel if both needed) and merge into one
|
|
1099
1192
|
let summary: string;
|
|
1100
1193
|
|
|
1194
|
+
// A single active Codex WebSocket session cannot service two concurrent
|
|
1195
|
+
// requests ("websocket request already in progress"). When the maintenance
|
|
1196
|
+
// calls use the Codex Responses provider, share one provider session, and
|
|
1197
|
+
// websocket transport is not explicitly disabled, run the split-turn history
|
|
1198
|
+
// and turn-prefix summaries sequentially. This covers websocket activation
|
|
1199
|
+
// from config/env/model defaults too: the provider can select websockets even
|
|
1200
|
+
// when `preferWebsockets` is undefined, while non-Codex providers keep the
|
|
1201
|
+
// previous parallel behavior.
|
|
1202
|
+
const summariesMayShareWebSocketSession = Boolean(
|
|
1203
|
+
model.api === "openai-codex-responses" &&
|
|
1204
|
+
summaryOptions.providerSessionState &&
|
|
1205
|
+
summaryOptions.preferWebsockets !== false,
|
|
1206
|
+
);
|
|
1207
|
+
|
|
1101
1208
|
if (isSplitTurn && turnPrefixMessages.length > 0) {
|
|
1102
|
-
|
|
1103
|
-
const [historyResult, turnPrefixResult] = await Promise.all([
|
|
1209
|
+
const runHistorySummary = () =>
|
|
1104
1210
|
messagesToSummarize.length > 0
|
|
1105
1211
|
? generateSummary(
|
|
1106
1212
|
messagesToSummarize,
|
|
@@ -1112,9 +1218,19 @@ export async function compact(
|
|
|
1112
1218
|
previousSummary,
|
|
1113
1219
|
summaryOptions,
|
|
1114
1220
|
)
|
|
1115
|
-
: Promise.resolve("No prior history.")
|
|
1116
|
-
|
|
1117
|
-
|
|
1221
|
+
: Promise.resolve("No prior history.");
|
|
1222
|
+
const runTurnPrefixSummary = () =>
|
|
1223
|
+
generateTurnPrefixSummary(turnPrefixMessages, model, settings.reserveTokens, apiKey, signal, summaryOptions);
|
|
1224
|
+
|
|
1225
|
+
let historyResult: string;
|
|
1226
|
+
let turnPrefixResult: string;
|
|
1227
|
+
if (summariesMayShareWebSocketSession) {
|
|
1228
|
+
// Sequential: avoids concurrent requests on the same provider session.
|
|
1229
|
+
historyResult = await runHistorySummary();
|
|
1230
|
+
turnPrefixResult = await runTurnPrefixSummary();
|
|
1231
|
+
} else {
|
|
1232
|
+
[historyResult, turnPrefixResult] = await Promise.all([runHistorySummary(), runTurnPrefixSummary()]);
|
|
1233
|
+
}
|
|
1118
1234
|
// Merge into single summary
|
|
1119
1235
|
summary = `${historyResult}\n\n---\n\n**Turn Context (split turn):**\n\n${turnPrefixResult}`;
|
|
1120
1236
|
} else if (messagesToSummarize.length > 0) {
|
|
@@ -1150,6 +1266,9 @@ export async function compact(
|
|
|
1150
1266
|
initiatorOverride: summaryOptions.initiatorOverride,
|
|
1151
1267
|
metadata: summaryOptions.metadata,
|
|
1152
1268
|
telemetry: summaryOptions.telemetry,
|
|
1269
|
+
sessionId: summaryOptions.sessionId,
|
|
1270
|
+
providerSessionState: summaryOptions.providerSessionState,
|
|
1271
|
+
preferWebsockets: summaryOptions.preferWebsockets,
|
|
1153
1272
|
},
|
|
1154
1273
|
);
|
|
1155
1274
|
|
|
@@ -1205,6 +1324,9 @@ async function generateTurnPrefixSummary(
|
|
|
1205
1324
|
reasoning: Effort.High,
|
|
1206
1325
|
initiatorOverride: options?.initiatorOverride,
|
|
1207
1326
|
metadata: options?.metadata,
|
|
1327
|
+
sessionId: options?.sessionId,
|
|
1328
|
+
providerSessionState: options?.providerSessionState,
|
|
1329
|
+
preferWebsockets: options?.preferWebsockets,
|
|
1208
1330
|
},
|
|
1209
1331
|
{ telemetry: options?.telemetry, oneshotKind: "compaction_turn_prefix" },
|
|
1210
1332
|
);
|