@gajae-code/agent-core 0.5.2 → 0.5.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,18 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.5.4] - 2026-06-17
6
+
7
+ ### Fixed
8
+
9
+ - Maintenance one-shot LLM calls now preserve active provider session state and the configured WebSocket transport preference. `SummaryOptions`, `HandoffOptions`, and `GenerateBranchSummaryOptions` accept `sessionId`, `providerSessionState`, and `preferWebsockets`, and `generateSummary`, `generateShortSummary`, `generateTurnPrefixSummary`, `generateHandoff`, `generateBranchSummary`, and `compact()` forward them through to `completeSimple` — previously these fields were dropped, so Codex/OpenAI-compatible compaction summaries, handoff generation, and branch summaries fell back to HTTP/SSE and lost `session_id` affinity even with `providers.openaiWebsockets: "on"`. Split-turn compaction now runs its history and turn-prefix summaries sequentially when they share a single provider WebSocket session, avoiding `websocket request already in progress`; non-WebSocket sessions still run them in parallel. `Agent` exposes a `preferWebsockets` getter so callers can forward the live transport preference (#736).
10
+
11
+ ## [0.5.3] - 2026-06-16
12
+
13
+ ### Fixed
14
+
15
+ - Bounded agent context growth, compaction, and token accounting for long-running sessions: `appendMessage` pushes in place instead of rebuilding the array; the append-only context keeps rolling per-message hashes instead of rescanning the full digest; an emergency-compaction floor that cannot be disabled now surfaces its reason; `getSessionStats` is single-pass; and `nativeCountTokens` skips the synchronous ~39 MB BPE tokenizer above a 2 MiB input cap, falling back to the cheap heuristic (#717).
16
+
5
17
  ## [0.5.2] - 2026-06-15
6
18
 
7
19
  ### Fixed
@@ -193,6 +193,12 @@ export declare class Agent {
193
193
  set sessionId(value: string | undefined);
194
194
  get providerSessionId(): string | undefined;
195
195
  set providerSessionId(value: string | undefined);
196
+ /**
197
+ * Whether websocket transport is preferred when the provider implementation
198
+ * supports it. Read by maintenance one-shot calls (compaction, handoff,
199
+ * branch summary) so they forward the same transport preference as live turns.
200
+ */
201
+ get preferWebsockets(): boolean | undefined;
196
202
  /**
197
203
  * Static metadata forwarded to every API request when no resolver is installed
198
204
  * (e.g. `metadata.user_id` for Anthropic session attribution). Setting this
@@ -4,7 +4,7 @@
4
4
  * When navigating to a different point in the session tree, this generates
5
5
  * a summary of the branch being left so context isn't lost.
6
6
  */
7
- import type { Model } from "@gajae-code/ai";
7
+ import type { Model, ProviderSessionState } from "@gajae-code/ai";
8
8
  import { type AgentTelemetry } from "../telemetry";
9
9
  import type { AgentMessage } from "../types";
10
10
  import type { ReadonlySessionManager, SessionEntry } from "./entries";
@@ -57,6 +57,15 @@ export interface GenerateBranchSummaryOptions {
57
57
  * wrapped in an OTEL chat span tagged with `pi.gen_ai.oneshot.kind = "branch_summary"`.
58
58
  */
59
59
  telemetry?: AgentTelemetry;
60
+ /**
61
+ * Provider session affinity id forwarded to the branch summary LLM call so it
62
+ * reuses the live turn's provider/WebSocket session.
63
+ */
64
+ sessionId?: string;
65
+ /** Shared provider state map so the branch summary call reuses session-scoped transport/session caches. */
66
+ providerSessionState?: Map<string, ProviderSessionState>;
67
+ /** Hint that websocket transport should be preferred when supported by the provider implementation. */
68
+ preferWebsockets?: boolean;
60
69
  }
61
70
  /**
62
71
  * Collect entries that should be summarized when navigating from one position to another.
@@ -4,7 +4,7 @@
4
4
  * Pure functions for compaction logic. The session manager handles I/O,
5
5
  * and after compaction the session is reloaded.
6
6
  */
7
- import { type MessageAttribution, type Model, type Usage } from "@gajae-code/ai";
7
+ import { type MessageAttribution, type Model, type ProviderSessionState, type Usage } from "@gajae-code/ai";
8
8
  import { type AgentTelemetry } from "../telemetry";
9
9
  import type { AgentMessage, AgentTool } from "../types";
10
10
  import type { SessionEntry } from "./entries";
@@ -66,6 +66,37 @@ export declare function effectiveReserveTokens(contextWindow: number, settings:
66
66
  * the safe input budget so prompt + reserved output cannot exceed the total window.
67
67
  */
68
68
  export declare function shouldCompact(contextTokens: number, contextWindow: number, settings: CompactionSettings, maxOutputTokens?: number): boolean;
69
+ /** Reason a compaction was triggered. `token` is the normal user-configurable path; the rest are emergency floors. */
70
+ export type CompactionTriggerReason = "token" | "heap" | "providerBytes" | "messageCount" | "imageBytes";
71
+ /** A point-in-time resource sample. Supplied by an injectable sampler so tests never read real RSS. */
72
+ export interface EmergencyCompactionSample {
73
+ /** Resident heap bytes (e.g. process.memoryUsage().heapUsed). */
74
+ heapUsedBytes: number;
75
+ /** Approximate serialized provider-context bytes. */
76
+ providerBytes: number;
77
+ /** Provider-visible message count. */
78
+ messageCount: number;
79
+ /** Approximate inline image bytes in the provider context. */
80
+ imageBytes: number;
81
+ }
82
+ export interface EmergencyCompactionLimits {
83
+ heapUsedBytes: number;
84
+ providerBytes: number;
85
+ messageCount: number;
86
+ imageBytes: number;
87
+ }
88
+ /**
89
+ * Non-disableable emergency floors. These sit well above normal usage and exist so a
90
+ * long session on weak hardware compacts before OOM even when token-based compaction is
91
+ * disabled or its threshold is set too high. They are NOT user-tunable down to zero.
92
+ */
93
+ export declare const DEFAULT_EMERGENCY_COMPACTION_LIMITS: EmergencyCompactionLimits;
94
+ /**
95
+ * Returns the first emergency limit exceeded (heap > providerBytes > imageBytes > messageCount),
96
+ * or null when none is. Pure and sampler-injected; the caller routes the result through the
97
+ * normal pair-safe `compact()` cut logic so a tool_use/tool_result pair is never split.
98
+ */
99
+ export declare function emergencyCompactionReason(sample: EmergencyCompactionSample, limits?: EmergencyCompactionLimits): CompactionTriggerReason | null;
69
100
  export declare function resolveThresholdTokens(contextWindow: number, settings: CompactionSettings, maxOutputTokens?: number): number;
70
101
  /**
71
102
  * Estimate token count for a message using the native o200k tokenizer.
@@ -151,6 +182,16 @@ export interface SummaryOptions {
151
182
  */
152
183
  telemetry?: AgentTelemetry;
153
184
  authCredentialType?: "api_key" | "oauth";
185
+ /**
186
+ * Provider session affinity id forwarded to the maintenance LLM call so it
187
+ * reuses the live turn's provider/WebSocket session (matches the
188
+ * `providerSessionId ?? sessionId` the agent loop sends for normal turns).
189
+ */
190
+ sessionId?: string;
191
+ /** Shared provider state map so maintenance calls reuse session-scoped transport/session caches. */
192
+ providerSessionState?: Map<string, ProviderSessionState>;
193
+ /** Hint that websocket transport should be preferred when supported by the provider implementation. */
194
+ preferWebsockets?: boolean;
154
195
  }
155
196
  export declare function generateSummary(currentMessages: AgentMessage[], model: Model, reserveTokens: number, apiKey: string, signal?: AbortSignal, customInstructions?: string, previousSummary?: string, options?: SummaryOptions): Promise<string>;
156
197
  export interface HandoffOptions {
@@ -168,6 +209,15 @@ export interface HandoffOptions {
168
209
  */
169
210
  telemetry?: AgentTelemetry;
170
211
  authCredentialType?: "api_key" | "oauth";
212
+ /**
213
+ * Provider session affinity id forwarded to the handoff LLM call so it
214
+ * reuses the live turn's provider/WebSocket session.
215
+ */
216
+ sessionId?: string;
217
+ /** Shared provider state map so the handoff call reuses session-scoped transport/session caches. */
218
+ providerSessionState?: Map<string, ProviderSessionState>;
219
+ /** Hint that websocket transport should be preferred when supported by the provider implementation. */
220
+ preferWebsockets?: boolean;
171
221
  }
172
222
  export declare function renderHandoffPrompt(customInstructions?: string): string;
173
223
  export declare function generateHandoff(messages: AgentMessage[], model: Model, apiKey: string, options: HandoffOptions, signal?: AbortSignal): Promise<string>;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/agent-core",
4
- "version": "0.5.2",
4
+ "version": "0.5.4",
5
5
  "description": "General-purpose agent with transport abstraction, state management, and attachment support",
6
6
  "homepage": "https://gaebal-gajae.dev",
7
7
  "author": "Yeachan-Heo",
@@ -35,9 +35,9 @@
35
35
  "fmt": "biome format --write ."
36
36
  },
37
37
  "dependencies": {
38
- "@gajae-code/ai": "0.5.2",
39
- "@gajae-code/natives": "0.5.2",
40
- "@gajae-code/utils": "0.5.2",
38
+ "@gajae-code/ai": "0.5.4",
39
+ "@gajae-code/natives": "0.5.4",
40
+ "@gajae-code/utils": "0.5.4",
41
41
  "@opentelemetry/api": "^1.9.0"
42
42
  },
43
43
  "devDependencies": {
package/src/agent.ts CHANGED
@@ -408,6 +408,15 @@ export class Agent {
408
408
  this.#providerSessionId = value;
409
409
  }
410
410
 
411
+ /**
412
+ * Whether websocket transport is preferred when the provider implementation
413
+ * supports it. Read by maintenance one-shot calls (compaction, handoff,
414
+ * branch summary) so they forward the same transport preference as live turns.
415
+ */
416
+ get preferWebsockets(): boolean | undefined {
417
+ return this.#preferWebsockets;
418
+ }
419
+
411
420
  /**
412
421
  * Static metadata forwarded to every API request when no resolver is installed
413
422
  * (e.g. `metadata.user_id` for Anthropic session attribution). Setting this
@@ -827,7 +836,10 @@ export class Agent {
827
836
  }
828
837
 
829
838
  appendMessage(m: AgentMessage) {
830
- this.#state.messages = [...this.#state.messages, m];
839
+ // In-place push (not [...messages, m]): appending M messages over a session of
840
+ // N is O(N+M), not O(M*N). Consumers read state.messages fresh; run() snapshots
841
+ // via slice() at the API boundary, so no caller relies on per-append array identity.
842
+ this.#state.messages.push(m);
831
843
  }
832
844
 
833
845
  popMessage(): AgentMessage | undefined {
@@ -199,8 +199,8 @@ export class AppendOnlyContextManager {
199
199
  readonly log = new AppendOnlyLog();
200
200
  /** How many normalized messages were synced into the log as of the last sync. */
201
201
  #lastSyncCount = 0;
202
- /** Fingerprint plus source bytes of synced message content — detects in-place rewrites with no hash-only equality. */
203
- #syncedDigest = emptyMessageDigest();
202
+ /** Per-synced-message content hashes (rolling digest). Detects in-place rewrites without retaining a full serialized-history string. */
203
+ #syncedHashes: (number | bigint)[] = [];
204
204
  /** Number of provider-normalized messages that were seeded before child-local messages. */
205
205
  #seededPrefixCount = 0;
206
206
 
@@ -236,35 +236,41 @@ export class AppendOnlyContextManager {
236
236
  const includesSeedPrefix =
237
237
  seededPrefixLength > 0 &&
238
238
  normalizedMessages.length >= seededPrefixLength &&
239
- this.#computeDigestRange(normalizedMessages, 0, seededPrefixLength).source ===
240
- this.#computeDigestRange(this.log.entries(), 0, seededPrefixLength).source;
239
+ this.#rangeHashesEqual(normalizedMessages, this.log.entries(), seededPrefixLength);
241
240
  const messagesToSync =
242
241
  seededPrefixLength > 0 && !includesSeedPrefix
243
242
  ? [...this.log.entries().slice(0, seededPrefixLength), ...normalizedMessages]
244
243
  : normalizedMessages;
245
244
 
246
- // Detect in-place rewrites of already-synced messages.
245
+ // Detect in-place rewrites of already-synced messages via per-message content
246
+ // hashes (no retained full serialized-history string; F5).
247
247
  if (
248
248
  this.#lastSyncCount > 0 &&
249
249
  this.#lastSyncCount <= messagesToSync.length &&
250
- this.#computeDigestRange(messagesToSync, 0, this.#lastSyncCount).source !== this.#syncedDigest.source
250
+ this.#prefixChanged(messagesToSync, this.#lastSyncCount)
251
251
  ) {
252
252
  if (this.#seededPrefixCount > 0) {
253
- throw new Error("AppendOnlyContextManager.syncMessages() seed prefix changed");
253
+ // F9: a seeded fork whose inherited prefix changed (e.g. after compaction)
254
+ // rebases onto the new provider context instead of throwing.
255
+ this.#rebaseToBaseline(normalizedMessages);
256
+ return;
254
257
  }
255
258
  this.log.clear();
256
259
  this.#lastSyncCount = 0;
260
+ this.#syncedHashes = [];
257
261
  }
258
262
 
259
- // Compaction — array shrunk. Seeded forks preserve the inherited prefix
260
- // and append child-local deltas, so a shorter child message array is not a
261
- // compaction signal while a seed prefix is active.
263
+ // Compaction — array shrunk. Seeded forks preserve the inherited prefix and
264
+ // append child-local deltas, so a shorter child array is not a compaction signal
265
+ // while a seed prefix is active; a genuine seeded compaction rebases (F9).
262
266
  if (messagesToSync.length < this.#lastSyncCount) {
263
267
  if (this.#seededPrefixCount > 0) {
264
- throw new Error("AppendOnlyContextManager.syncMessages() cannot compact a seeded fork without reset");
268
+ this.#rebaseToBaseline(normalizedMessages);
269
+ return;
265
270
  }
266
271
  this.log.clear();
267
272
  this.#lastSyncCount = 0;
273
+ this.#syncedHashes = [];
268
274
  }
269
275
 
270
276
  const newMsgs = messagesToSync.slice(this.#lastSyncCount);
@@ -273,7 +279,7 @@ export class AppendOnlyContextManager {
273
279
  }
274
280
 
275
281
  this.#lastSyncCount = messagesToSync.length;
276
- this.#syncedDigest = this.#computeDigest(messagesToSync);
282
+ this.#syncedHashes = this.#hashRange(messagesToSync, 0, messagesToSync.length);
277
283
  }
278
284
 
279
285
  seedNormalizedMessages(messages: readonly Message[], options?: { reset?: boolean }): void {
@@ -284,7 +290,7 @@ export class AppendOnlyContextManager {
284
290
  this.log.clear();
285
291
  this.log.extend(clonedMessages);
286
292
  this.#lastSyncCount = clonedMessages.length;
287
- this.#syncedDigest = this.#computeDigest(clonedMessages);
293
+ this.#syncedHashes = this.#hashRange(clonedMessages, 0, clonedMessages.length);
288
294
  this.#seededPrefixCount = clonedMessages.length;
289
295
  }
290
296
 
@@ -293,7 +299,7 @@ export class AppendOnlyContextManager {
293
299
  this.prefix.invalidate();
294
300
  this.log.clear();
295
301
  this.#lastSyncCount = 0;
296
- this.#syncedDigest = emptyMessageDigest();
302
+ this.#syncedHashes = [];
297
303
  this.#seededPrefixCount = 0;
298
304
  }
299
305
 
@@ -301,7 +307,7 @@ export class AppendOnlyContextManager {
301
307
  resetSyncCursor(): void {
302
308
  this.log.clear();
303
309
  this.#lastSyncCount = 0;
304
- this.#syncedDigest = emptyMessageDigest();
310
+ this.#syncedHashes = [];
305
311
  this.#seededPrefixCount = 0;
306
312
  }
307
313
 
@@ -321,28 +327,45 @@ export class AppendOnlyContextManager {
321
327
  this.prefix.invalidate();
322
328
  this.log.clear();
323
329
  this.#lastSyncCount = 0;
324
- this.#syncedDigest = emptyMessageDigest();
330
+ this.#syncedHashes = [];
325
331
  this.#seededPrefixCount = 0;
326
332
  this.prefix.build(context, options);
327
333
  }
328
334
 
329
- /**
330
- * Deterministic digest over the provider-visible message payload. The source
331
- * string is kept and compared for equality so the hash is only a fast summary,
332
- * never the authority for accepting append-only sync state.
333
- */
334
- #computeDigest(messages: readonly unknown[]): MessageDigest {
335
- return this.#computeDigestRange(messages, 0, messages.length);
335
+ #hashMessage(message: unknown): number | bigint {
336
+ return hashSource(JSON.stringify(message) ?? "null");
337
+ }
338
+
339
+ #hashRange(messages: readonly unknown[], start: number, end: number): (number | bigint)[] {
340
+ const out: (number | bigint)[] = [];
341
+ for (let i = start; i < end; i++) out.push(this.#hashMessage(messages[i]));
342
+ return out;
343
+ }
344
+
345
+ /** True when the first `count` messages of `a` and `b` are content-equal by per-message hash. */
346
+ #rangeHashesEqual(a: readonly unknown[], b: readonly unknown[], count: number): boolean {
347
+ for (let i = 0; i < count; i++) {
348
+ if (this.#hashMessage(a[i]) !== this.#hashMessage(b[i])) return false;
349
+ }
350
+ return true;
336
351
  }
337
352
 
338
- #computeDigestRange(messages: readonly unknown[], start: number, end: number): MessageDigest {
339
- let source = "[";
340
- for (let i = start; i < end; i++) {
341
- if (i > start) source += ",";
342
- source += JSON.stringify(messages[i]) ?? "null";
353
+ /** True when any of the first `count` already-synced messages changed content (in-place rewrite). */
354
+ #prefixChanged(messages: readonly unknown[], count: number): boolean {
355
+ if (count > this.#syncedHashes.length) return false;
356
+ for (let i = 0; i < count; i++) {
357
+ if (this.#hashMessage(messages[i]) !== this.#syncedHashes[i]) return true;
343
358
  }
344
- source += "]";
345
- return { hash: hashSource(source), source };
359
+ return false;
360
+ }
361
+
362
+ /** F9: reset the seeded log to a new provider-visible baseline (seeded compaction/rebase). */
363
+ #rebaseToBaseline(messages: readonly unknown[]): void {
364
+ this.log.clear();
365
+ this.log.extend([...messages]);
366
+ this.#lastSyncCount = messages.length;
367
+ this.#seededPrefixCount = 0;
368
+ this.#syncedHashes = this.#hashRange(messages, 0, messages.length);
346
369
  }
347
370
  }
348
371
 
@@ -350,15 +373,6 @@ export class AppendOnlyContextManager {
350
373
  // Snapshot helpers
351
374
  // ---------------------------------------------------------------------------
352
375
 
353
- type MessageDigest = {
354
- hash: number | bigint;
355
- source: string;
356
- };
357
-
358
- function emptyMessageDigest(): MessageDigest {
359
- return { hash: hashSource("[]"), source: "[]" };
360
- }
361
-
362
376
  function hashSource(source: string): number | bigint {
363
377
  return typeof Bun !== "undefined" ? Bun.hash(source) : hashString32(source);
364
378
  }
@@ -5,7 +5,7 @@
5
5
  * a summary of the branch being left so context isn't lost.
6
6
  */
7
7
 
8
- import type { Model } from "@gajae-code/ai";
8
+ import type { Model, ProviderSessionState } from "@gajae-code/ai";
9
9
  import { prompt } from "@gajae-code/utils";
10
10
  import { type AgentTelemetry, instrumentedCompleteSimple } from "../telemetry";
11
11
  import type { AgentMessage } from "../types";
@@ -86,6 +86,15 @@ export interface GenerateBranchSummaryOptions {
86
86
  * wrapped in an OTEL chat span tagged with `pi.gen_ai.oneshot.kind = "branch_summary"`.
87
87
  */
88
88
  telemetry?: AgentTelemetry;
89
+ /**
90
+ * Provider session affinity id forwarded to the branch summary LLM call so it
91
+ * reuses the live turn's provider/WebSocket session.
92
+ */
93
+ sessionId?: string;
94
+ /** Shared provider state map so the branch summary call reuses session-scoped transport/session caches. */
95
+ providerSessionState?: Map<string, ProviderSessionState>;
96
+ /** Hint that websocket transport should be preferred when supported by the provider implementation. */
97
+ preferWebsockets?: boolean;
89
98
  }
90
99
 
91
100
  // ============================================================================
@@ -274,7 +283,17 @@ export async function generateBranchSummary(
274
283
  entries: SessionEntry[],
275
284
  options: GenerateBranchSummaryOptions,
276
285
  ): Promise<BranchSummaryResult> {
277
- const { model, apiKey, signal, customInstructions, reserveTokens = 16384, metadata } = options;
286
+ const {
287
+ model,
288
+ apiKey,
289
+ signal,
290
+ customInstructions,
291
+ reserveTokens = 16384,
292
+ metadata,
293
+ sessionId,
294
+ providerSessionState,
295
+ preferWebsockets,
296
+ } = options;
278
297
 
279
298
  // Token budget = context window minus reserved space for prompt + response
280
299
  const contextWindow = model.contextWindow || 128000;
@@ -307,7 +326,7 @@ export async function generateBranchSummary(
307
326
  const response = await instrumentedCompleteSimple(
308
327
  model,
309
328
  { systemPrompt: [SUMMARIZATION_SYSTEM_PROMPT], messages: summarizationMessages },
310
- { apiKey, signal, maxTokens: 2048, metadata },
329
+ { apiKey, signal, maxTokens: 2048, metadata, sessionId, providerSessionState, preferWebsockets },
311
330
  { telemetry: options.telemetry, oneshotKind: "branch_summary" },
312
331
  );
313
332
 
@@ -12,6 +12,7 @@ import {
12
12
  type Message,
13
13
  type MessageAttribution,
14
14
  type Model,
15
+ type ProviderSessionState,
15
16
  type Usage,
16
17
  } from "@gajae-code/ai";
17
18
  import { isCompiledBinary, logger, prompt } from "@gajae-code/utils";
@@ -236,6 +237,56 @@ export function shouldCompact(
236
237
  return contextTokens > thresholdTokens;
237
238
  }
238
239
 
240
+ /** Reason a compaction was triggered. `token` is the normal user-configurable path; the rest are emergency floors. */
241
+ export type CompactionTriggerReason = "token" | "heap" | "providerBytes" | "messageCount" | "imageBytes";
242
+
243
+ /** A point-in-time resource sample. Supplied by an injectable sampler so tests never read real RSS. */
244
+ export interface EmergencyCompactionSample {
245
+ /** Resident heap bytes (e.g. process.memoryUsage().heapUsed). */
246
+ heapUsedBytes: number;
247
+ /** Approximate serialized provider-context bytes. */
248
+ providerBytes: number;
249
+ /** Provider-visible message count. */
250
+ messageCount: number;
251
+ /** Approximate inline image bytes in the provider context. */
252
+ imageBytes: number;
253
+ }
254
+
255
+ export interface EmergencyCompactionLimits {
256
+ heapUsedBytes: number;
257
+ providerBytes: number;
258
+ messageCount: number;
259
+ imageBytes: number;
260
+ }
261
+
262
+ /**
263
+ * Non-disableable emergency floors. These sit well above normal usage and exist so a
264
+ * long session on weak hardware compacts before OOM even when token-based compaction is
265
+ * disabled or its threshold is set too high. They are NOT user-tunable down to zero.
266
+ */
267
+ export const DEFAULT_EMERGENCY_COMPACTION_LIMITS: EmergencyCompactionLimits = {
268
+ heapUsedBytes: 1_536 * 1024 * 1024, // 1.5 GiB resident heap
269
+ providerBytes: 24 * 1024 * 1024, // 24 MiB serialized provider context
270
+ messageCount: 4000,
271
+ imageBytes: 64 * 1024 * 1024, // 64 MiB inline image bytes
272
+ };
273
+
274
+ /**
275
+ * Returns the first emergency limit exceeded (heap > providerBytes > imageBytes > messageCount),
276
+ * or null when none is. Pure and sampler-injected; the caller routes the result through the
277
+ * normal pair-safe `compact()` cut logic so a tool_use/tool_result pair is never split.
278
+ */
279
+ export function emergencyCompactionReason(
280
+ sample: EmergencyCompactionSample,
281
+ limits: EmergencyCompactionLimits = DEFAULT_EMERGENCY_COMPACTION_LIMITS,
282
+ ): CompactionTriggerReason | null {
283
+ if (sample.heapUsedBytes > limits.heapUsedBytes) return "heap";
284
+ if (sample.providerBytes > limits.providerBytes) return "providerBytes";
285
+ if (sample.imageBytes > limits.imageBytes) return "imageBytes";
286
+ if (sample.messageCount > limits.messageCount) return "messageCount";
287
+ return null;
288
+ }
289
+
239
290
  export function resolveThresholdTokens(
240
291
  contextWindow: number,
241
292
  settings: CompactionSettings,
@@ -301,7 +352,18 @@ function nativeTokenizerEntrypoint(): string {
301
352
  return isCompiledBinary() ? COMPILED_NATIVE_TOKENIZER_ENTRYPOINT : SOURCE_NATIVE_TOKENIZER_ENTRYPOINT;
302
353
  }
303
354
 
355
+ /** Max total fragment chars sent to the synchronous native tokenizer (F22). */
356
+ const MAX_NATIVE_TOKENIZE_CHARS = 2 * 1024 * 1024;
357
+
304
358
  function nativeCountTokens(fragments: string[]): number {
359
+ let totalChars = 0;
360
+ for (const fragment of fragments) totalChars += fragment.length;
361
+ if (totalChars > MAX_NATIVE_TOKENIZE_CHARS) {
362
+ // F22: skip the synchronous native BPE tokenizer (materializes a ~39MB table and is
363
+ // O(text)) on pathologically large inputs; the cheap chars/token heuristic is more
364
+ // than accurate enough for size/budget decisions and never blocks the event loop.
365
+ return estimateTextTokensHeuristic(fragments);
366
+ }
305
367
  if (!cachedNativeCountTokens) {
306
368
  const natives = requireFromCompaction(nativeTokenizerEntrypoint()) as NativeTokenizerModule;
307
369
  cachedNativeCountTokens = natives.countTokens;
@@ -674,6 +736,16 @@ export interface SummaryOptions {
674
736
  */
675
737
  telemetry?: AgentTelemetry;
676
738
  authCredentialType?: "api_key" | "oauth";
739
+ /**
740
+ * Provider session affinity id forwarded to the maintenance LLM call so it
741
+ * reuses the live turn's provider/WebSocket session (matches the
742
+ * `providerSessionId ?? sessionId` the agent loop sends for normal turns).
743
+ */
744
+ sessionId?: string;
745
+ /** Shared provider state map so maintenance calls reuse session-scoped transport/session caches. */
746
+ providerSessionState?: Map<string, ProviderSessionState>;
747
+ /** Hint that websocket transport should be preferred when supported by the provider implementation. */
748
+ preferWebsockets?: boolean;
677
749
  }
678
750
 
679
751
  export async function generateSummary(
@@ -740,6 +812,9 @@ export async function generateSummary(
740
812
  reasoning: Effort.High,
741
813
  initiatorOverride: options?.initiatorOverride,
742
814
  metadata: options?.metadata,
815
+ sessionId: options?.sessionId,
816
+ providerSessionState: options?.providerSessionState,
817
+ preferWebsockets: options?.preferWebsockets,
743
818
  },
744
819
  { telemetry: options?.telemetry, oneshotKind: "compaction_summary" },
745
820
  );
@@ -775,6 +850,15 @@ export interface HandoffOptions {
775
850
  */
776
851
  telemetry?: AgentTelemetry;
777
852
  authCredentialType?: "api_key" | "oauth";
853
+ /**
854
+ * Provider session affinity id forwarded to the handoff LLM call so it
855
+ * reuses the live turn's provider/WebSocket session.
856
+ */
857
+ sessionId?: string;
858
+ /** Shared provider state map so the handoff call reuses session-scoped transport/session caches. */
859
+ providerSessionState?: Map<string, ProviderSessionState>;
860
+ /** Hint that websocket transport should be preferred when supported by the provider implementation. */
861
+ preferWebsockets?: boolean;
778
862
  }
779
863
 
780
864
  export function renderHandoffPrompt(customInstructions?: string): string {
@@ -816,6 +900,9 @@ export async function generateHandoff(
816
900
  toolChoice: "none",
817
901
  initiatorOverride: options.initiatorOverride,
818
902
  metadata: options.metadata,
903
+ sessionId: options.sessionId,
904
+ providerSessionState: options.providerSessionState,
905
+ preferWebsockets: options.preferWebsockets,
819
906
  },
820
907
  { telemetry: options.telemetry, oneshotKind: "handoff" },
821
908
  );
@@ -875,6 +962,9 @@ async function generateShortSummary(
875
962
  reasoning: Effort.High,
876
963
  initiatorOverride: options?.initiatorOverride,
877
964
  metadata: options?.metadata,
965
+ sessionId: options?.sessionId,
966
+ providerSessionState: options?.providerSessionState,
967
+ preferWebsockets: options?.preferWebsockets,
878
968
  },
879
969
  { telemetry: options?.telemetry, oneshotKind: "compaction_short_summary" },
880
970
  );
@@ -1059,6 +1149,9 @@ export async function compact(
1059
1149
  metadata: options?.metadata,
1060
1150
  convertToLlm: options?.convertToLlm,
1061
1151
  telemetry: options?.telemetry,
1152
+ sessionId: options?.sessionId,
1153
+ providerSessionState: options?.providerSessionState,
1154
+ preferWebsockets: options?.preferWebsockets,
1062
1155
  };
1063
1156
 
1064
1157
  let preserveData = withOpenAiRemoteCompactionPreserveData(previousPreserveData, undefined);
@@ -1098,9 +1191,22 @@ export async function compact(
1098
1191
  // Generate summaries (can be parallel if both needed) and merge into one
1099
1192
  let summary: string;
1100
1193
 
1194
+ // A single active Codex WebSocket session cannot service two concurrent
1195
+ // requests ("websocket request already in progress"). When the maintenance
1196
+ // calls use the Codex Responses provider, share one provider session, and
1197
+ // websocket transport is not explicitly disabled, run the split-turn history
1198
+ // and turn-prefix summaries sequentially. This covers websocket activation
1199
+ // from config/env/model defaults too: the provider can select websockets even
1200
+ // when `preferWebsockets` is undefined, while non-Codex providers keep the
1201
+ // previous parallel behavior.
1202
+ const summariesMayShareWebSocketSession = Boolean(
1203
+ model.api === "openai-codex-responses" &&
1204
+ summaryOptions.providerSessionState &&
1205
+ summaryOptions.preferWebsockets !== false,
1206
+ );
1207
+
1101
1208
  if (isSplitTurn && turnPrefixMessages.length > 0) {
1102
- // Generate both summaries in parallel
1103
- const [historyResult, turnPrefixResult] = await Promise.all([
1209
+ const runHistorySummary = () =>
1104
1210
  messagesToSummarize.length > 0
1105
1211
  ? generateSummary(
1106
1212
  messagesToSummarize,
@@ -1112,9 +1218,19 @@ export async function compact(
1112
1218
  previousSummary,
1113
1219
  summaryOptions,
1114
1220
  )
1115
- : Promise.resolve("No prior history."),
1116
- generateTurnPrefixSummary(turnPrefixMessages, model, settings.reserveTokens, apiKey, signal, summaryOptions),
1117
- ]);
1221
+ : Promise.resolve("No prior history.");
1222
+ const runTurnPrefixSummary = () =>
1223
+ generateTurnPrefixSummary(turnPrefixMessages, model, settings.reserveTokens, apiKey, signal, summaryOptions);
1224
+
1225
+ let historyResult: string;
1226
+ let turnPrefixResult: string;
1227
+ if (summariesMayShareWebSocketSession) {
1228
+ // Sequential: avoids concurrent requests on the same provider session.
1229
+ historyResult = await runHistorySummary();
1230
+ turnPrefixResult = await runTurnPrefixSummary();
1231
+ } else {
1232
+ [historyResult, turnPrefixResult] = await Promise.all([runHistorySummary(), runTurnPrefixSummary()]);
1233
+ }
1118
1234
  // Merge into single summary
1119
1235
  summary = `${historyResult}\n\n---\n\n**Turn Context (split turn):**\n\n${turnPrefixResult}`;
1120
1236
  } else if (messagesToSummarize.length > 0) {
@@ -1150,6 +1266,9 @@ export async function compact(
1150
1266
  initiatorOverride: summaryOptions.initiatorOverride,
1151
1267
  metadata: summaryOptions.metadata,
1152
1268
  telemetry: summaryOptions.telemetry,
1269
+ sessionId: summaryOptions.sessionId,
1270
+ providerSessionState: summaryOptions.providerSessionState,
1271
+ preferWebsockets: summaryOptions.preferWebsockets,
1153
1272
  },
1154
1273
  );
1155
1274
 
@@ -1205,6 +1324,9 @@ async function generateTurnPrefixSummary(
1205
1324
  reasoning: Effort.High,
1206
1325
  initiatorOverride: options?.initiatorOverride,
1207
1326
  metadata: options?.metadata,
1327
+ sessionId: options?.sessionId,
1328
+ providerSessionState: options?.providerSessionState,
1329
+ preferWebsockets: options?.preferWebsockets,
1208
1330
  },
1209
1331
  { telemetry: options?.telemetry, oneshotKind: "compaction_turn_prefix" },
1210
1332
  );