@gajae-code/ai 0.5.2 → 0.5.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,18 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.5.3] - 2026-06-16
6
+
7
+ ### Added
8
+
9
+ - Added opt-in `AuthStorageOptions.credentialRankingMode` (`balanced` (default) | `earliest-reset`) for multi-account OAuth credential selection. `earliest-reset` ranks non-blocked credentials earliest-expiry-first — draining the soonest-to-reset account before its perishable tumbling-window quota (e.g. Claude 5h/7d) is lost at reset — keeping the existing drain-rate/used-fraction metrics as tiebreakers. `balanced` is byte-identical to prior behavior, and ranking only runs at session start (or when the session's preferred credential is blocked), so this never thrashes accounts mid-session.
10
+
11
+ ### Fixed
12
+
13
+ - Allowed `openai-codex-responses` custom backends to use opaque `apiKey` bearer tokens by omitting `chatgpt-account-id` when the token does not expose a Codex account id.
14
+ - Fixed OpenAI code websocket continuations to treat codex-lb's `codex_previous_response_stale` response failures as expired `previous_response_id` anchors and retry with full context instead of surfacing the transient failure.
15
+ - Bounded the Cursor provider's conversation cache with an LRU(64) + 1h TTL and added `disposeCursorConversation`, so long-running sessions no longer retain Cursor conversation state without limit (#717).
16
+
5
17
  ## [0.5.2] - 2026-06-15
6
18
 
7
19
  ### Changed
@@ -229,9 +229,26 @@ export interface CredentialDisabledEvent {
229
229
  provider: string;
230
230
  disabledCause: string;
231
231
  }
232
+ /**
233
+ * How {@link AuthStorage} orders multiple healthy OAuth credentials of the same
234
+ * provider:type pool when selecting one for a (new) session.
235
+ *
236
+ * - `balanced` (default): prefer the least-used / lowest-drain-rate account.
237
+ * Spreads load across accounts and keeps burst headroom on every account.
238
+ * - `earliest-reset`: prefer the non-blocked account whose usage window resets
239
+ * soonest (earliest-expiry-first). Tumbling-window quota is perishable —
240
+ * unused quota is lost at reset — so draining the soonest-to-reset account
241
+ * first minimizes wasted quota. Drain/used metrics remain tiebreakers.
242
+ *
243
+ * Only affects ranking, which the `shouldRank` guard already limits to session
244
+ * start (or when the session's preferred credential is blocked), so this never
245
+ * thrashes accounts mid-session / cold-starts the server-side prompt cache.
246
+ */
247
+ export type CredentialRankingMode = "balanced" | "earliest-reset";
232
248
  export type AuthStorageOptions = {
233
249
  usageProviderResolver?: (provider: Provider) => UsageProvider | undefined;
234
250
  rankingStrategyResolver?: (provider: Provider) => CredentialRankingStrategy | undefined;
251
+ credentialRankingMode?: CredentialRankingMode;
235
252
  usageFetch?: typeof fetch;
236
253
  usageRequestTimeoutMs?: number;
237
254
  usageLogger?: UsageLogger;
@@ -2,6 +2,8 @@ import { type JsonValue } from "@bufbuild/protobuf";
2
2
  import type { CursorExecHandlerResult, CursorExecHandlers, CursorToolResultHandler, Message, StreamFunction, StreamOptions, ToolResultMessage } from "../types";
3
3
  export declare const CURSOR_API_URL = "https://api2.cursor.sh";
4
4
  export declare const CURSOR_CLIENT_VERSION = "cli-2026.01.09-231024f";
5
+ /** Drop all cached state + blob bytes for a conversation (F15 bound + session-teardown hook). */
6
+ export declare function disposeCursorConversation(conversationId: string): void;
5
7
  export interface CursorOptions extends StreamOptions {
6
8
  customSystemPrompt?: string;
7
9
  conversationId?: string;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/ai",
4
- "version": "0.5.2",
4
+ "version": "0.5.3",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://gaebal-gajae.dev",
7
7
  "author": "Yeachan-Heo",
@@ -43,7 +43,7 @@
43
43
  "dependencies": {
44
44
  "@anthropic-ai/sdk": "^0.94.0",
45
45
  "@bufbuild/protobuf": "^2.12.0",
46
- "@gajae-code/utils": "0.5.2",
46
+ "@gajae-code/utils": "0.5.3",
47
47
  "openai": "^6.36.0",
48
48
  "partial-json": "^0.1.7",
49
49
  "zod": "4.4.3"
@@ -288,9 +288,27 @@ export interface CredentialDisabledEvent {
288
288
  disabledCause: string;
289
289
  }
290
290
 
291
+ /**
292
+ * How {@link AuthStorage} orders multiple healthy OAuth credentials of the same
293
+ * provider:type pool when selecting one for a (new) session.
294
+ *
295
+ * - `balanced` (default): prefer the least-used / lowest-drain-rate account.
296
+ * Spreads load across accounts and keeps burst headroom on every account.
297
+ * - `earliest-reset`: prefer the non-blocked account whose usage window resets
298
+ * soonest (earliest-expiry-first). Tumbling-window quota is perishable —
299
+ * unused quota is lost at reset — so draining the soonest-to-reset account
300
+ * first minimizes wasted quota. Drain/used metrics remain tiebreakers.
301
+ *
302
+ * Only affects ranking, which the `shouldRank` guard already limits to session
303
+ * start (or when the session's preferred credential is blocked), so this never
304
+ * thrashes accounts mid-session / cold-starts the server-side prompt cache.
305
+ */
306
+ export type CredentialRankingMode = "balanced" | "earliest-reset";
307
+
291
308
  export type AuthStorageOptions = {
292
309
  usageProviderResolver?: (provider: Provider) => UsageProvider | undefined;
293
310
  rankingStrategyResolver?: (provider: Provider) => CredentialRankingStrategy | undefined;
311
+ credentialRankingMode?: CredentialRankingMode;
294
312
  usageFetch?: typeof fetch;
295
313
  usageRequestTimeoutMs?: number;
296
314
  usageLogger?: UsageLogger;
@@ -655,6 +673,7 @@ export class AuthStorage {
655
673
  #usageReportsInFlight: Map<string, Promise<UsageReport[] | null>> = new Map();
656
674
  #usageFetch: typeof fetch;
657
675
  #usageRequestTimeoutMs: number;
676
+ #credentialRankingMode: CredentialRankingMode = "balanced";
658
677
  #usageLogger?: UsageLogger;
659
678
  #fallbackResolver?: (provider: string) => string | undefined;
660
679
  #store: AuthCredentialStore;
@@ -686,6 +705,7 @@ export class AuthStorage {
686
705
  this.#usageCache = new AuthStorageUsageCache(this.#store);
687
706
  this.#usageFetch = options.usageFetch ?? fetch;
688
707
  this.#usageRequestTimeoutMs = options.usageRequestTimeoutMs ?? DEFAULT_USAGE_REQUEST_TIMEOUT_MS;
708
+ this.#credentialRankingMode = options.credentialRankingMode ?? "balanced";
689
709
  this.#refreshOAuthCredentialOverride = options.refreshOAuthCredential;
690
710
  this.#fetchUsageReportsOverride = options.fetchUsageReports;
691
711
  this.#sourceLabel = options.sourceLabel;
@@ -2453,6 +2473,7 @@ export class AuthStorage {
2453
2473
  secondaryDrainRate: number;
2454
2474
  primaryUsed: number;
2455
2475
  primaryDrainRate: number;
2476
+ resetAtMs: number;
2456
2477
  orderPos: number;
2457
2478
  }> = [];
2458
2479
  // Pre-fetch usage reports in parallel for non-blocked credentials.
@@ -2522,6 +2543,10 @@ export class AuthStorage {
2522
2543
  ),
2523
2544
  primaryUsed: this.#normalizeUsageFraction(primary),
2524
2545
  primaryDrainRate: this.#computeWindowDrainRate(primary, nowMs, strategy.windowDefaults.primaryMs),
2546
+ resetAtMs:
2547
+ this.#resolveWindowResetAt(primary?.window) ??
2548
+ this.#resolveWindowResetAt(secondary?.window) ??
2549
+ Number.POSITIVE_INFINITY,
2525
2550
  orderPos,
2526
2551
  });
2527
2552
  }
@@ -2539,6 +2564,11 @@ export class AuthStorage {
2539
2564
  if (leftPlanPriority !== rightPlanPriority) return leftPlanPriority - rightPlanPriority;
2540
2565
  }
2541
2566
  if (left.hasPriorityBoost !== right.hasPriorityBoost) return left.hasPriorityBoost ? -1 : 1;
2567
+ if (this.#credentialRankingMode === "earliest-reset" && left.resetAtMs !== right.resetAtMs) {
2568
+ // Earliest-expiry-first: drain the soonest-to-reset account before
2569
+ // its perishable tumbling-window quota is lost at reset.
2570
+ return left.resetAtMs - right.resetAtMs;
2571
+ }
2542
2572
  if (left.secondaryDrainRate !== right.secondaryDrainRate)
2543
2573
  return left.secondaryDrainRate - right.secondaryDrainRate;
2544
2574
  if (left.secondaryUsed !== right.secondaryUsed) return left.secondaryUsed - right.secondaryUsed;
@@ -136,6 +136,38 @@ export const CURSOR_CLIENT_VERSION = "cli-2026.01.09-231024f";
136
136
  const conversationStateCache = new Map<string, ConversationStateStructure>();
137
137
  const conversationBlobStores = new Map<string, Map<string, Uint8Array>>();
138
138
 
139
+ // F15: bound the module-global conversation caches so long-lived / many-session use cannot
140
+ // grow them without limit. LRU by conversation count + TTL on idle conversations.
141
+ const CURSOR_MAX_CONVERSATIONS = 64;
142
+ const CURSOR_CONVERSATION_TTL_MS = 60 * 60 * 1000;
143
+ const conversationLastAccess = new Map<string, number>();
144
+
145
+ /** Drop all cached state + blob bytes for a conversation (F15 bound + session-teardown hook). */
146
+ export function disposeCursorConversation(conversationId: string): void {
147
+ conversationStateCache.delete(conversationId);
148
+ conversationBlobStores.delete(conversationId);
149
+ conversationLastAccess.delete(conversationId);
150
+ }
151
+
152
+ /** Refresh recency for a conversation and evict TTL-stale / LRU-overflow entries (F15). */
153
+ function touchCursorConversation(conversationId: string): void {
154
+ const now = Date.now();
155
+ for (const [id, ts] of conversationLastAccess) {
156
+ if (id !== conversationId && now - ts > CURSOR_CONVERSATION_TTL_MS) disposeCursorConversation(id);
157
+ }
158
+ conversationLastAccess.set(conversationId, now);
159
+ const state = conversationStateCache.get(conversationId);
160
+ if (state !== undefined) {
161
+ conversationStateCache.delete(conversationId);
162
+ conversationStateCache.set(conversationId, state);
163
+ }
164
+ while (conversationStateCache.size > CURSOR_MAX_CONVERSATIONS) {
165
+ const oldest = conversationStateCache.keys().next().value;
166
+ if (oldest === undefined || oldest === conversationId) break;
167
+ disposeCursorConversation(oldest);
168
+ }
169
+ }
170
+
139
171
  export interface CursorOptions extends StreamOptions {
140
172
  customSystemPrompt?: string;
141
173
  conversationId?: string;
@@ -349,6 +381,7 @@ export const streamCursor: StreamFunction<"cursor-agent"> = (
349
381
  conversationState: cachedState,
350
382
  });
351
383
  conversationStateCache.set(conversationId, conversationState);
384
+ touchCursorConversation(conversationId);
352
385
  const requestContextTools = buildMcpToolDefinitions(context.tools);
353
386
 
354
387
  const baseUrl = model.baseUrl || CURSOR_API_URL;
@@ -405,6 +438,7 @@ export const streamCursor: StreamFunction<"cursor-agent"> = (
405
438
 
406
439
  const onConversationCheckpoint = (checkpoint: ConversationStateStructure) => {
407
440
  conversationStateCache.set(conversationId, checkpoint);
441
+ touchCursorConversation(conversationId);
408
442
  };
409
443
 
410
444
  let resolveH2: (() => void) | undefined;
@@ -96,6 +96,7 @@ const CODEX_WEBSOCKET_IDLE_TIMEOUT_MS = 300000;
96
96
  const CODEX_WEBSOCKET_FIRST_EVENT_TIMEOUT_MS = 15000;
97
97
  const CODEX_WEBSOCKET_RETRY_BUDGET = CODEX_MAX_RETRIES;
98
98
  const CODEX_WEBSOCKET_TRANSPORT_ERROR_PREFIX = "Codex websocket transport error";
99
+ const CODEX_PREVIOUS_RESPONSE_STALE_CODES = new Set(["previous_response_not_found", "codex_previous_response_stale"]);
99
100
  const CODEX_RETRYABLE_EVENT_CODES = new Set(["model_error", "server_error", "internal_error"]);
100
101
  const CODEX_RETRYABLE_EVENT_MESSAGE =
101
102
  /processing your request|retry your request|temporar(?:y|ily)|overloaded|service.?unavailable|internal error|server error/i;
@@ -170,7 +171,7 @@ interface CodexProviderSessionState extends ProviderSessionState {
170
171
 
171
172
  interface CodexRequestContext {
172
173
  apiKey: string;
173
- accountId: string;
174
+ accountId: string | undefined;
174
175
  baseUrl: string;
175
176
  url: string;
176
177
  requestHeaders: Record<string, string>;
@@ -1475,7 +1476,11 @@ async function tryReconnectCodexWebSocketOnConnectionLimit(
1475
1476
  }
1476
1477
 
1477
1478
  function isCodexPreviousResponseNotFound(error: unknown): boolean {
1478
- return error instanceof CodexProviderStreamError && error.code === "previous_response_not_found";
1479
+ return (
1480
+ error instanceof CodexProviderStreamError &&
1481
+ typeof error.code === "string" &&
1482
+ CODEX_PREVIOUS_RESPONSE_STALE_CODES.has(error.code)
1483
+ );
1479
1484
  }
1480
1485
 
1481
1486
  async function tryRecoverCodexPreviousResponseNotFound(
@@ -1801,12 +1806,12 @@ export async function prewarmOpenAICodexResponses(
1801
1806
  function getCodexWebSocketSessionKey(
1802
1807
  sessionId: string | undefined,
1803
1808
  model: Model<"openai-codex-responses">,
1804
- accountId: string,
1809
+ accountId: string | undefined,
1805
1810
  baseUrl: string,
1806
1811
  ): string | undefined {
1807
1812
  const promptCacheKey = normalizeOpenAIResponsesPromptCacheKey(sessionId);
1808
1813
  if (!promptCacheKey) return undefined;
1809
- return `${accountId}:${baseUrl}:${model.id}:${promptCacheKey}`;
1814
+ return `${accountId ?? "opaque"}:${baseUrl}:${model.id}:${promptCacheKey}`;
1810
1815
  }
1811
1816
 
1812
1817
  function getCodexPublicSessionKey(
@@ -2335,7 +2340,7 @@ async function getOrCreateCodexWebSocketConnection(
2335
2340
  async function openCodexSseEventStream(
2336
2341
  url: string,
2337
2342
  requestHeaders: Record<string, string> | undefined,
2338
- accountId: string,
2343
+ accountId: string | undefined,
2339
2344
  apiKey: string,
2340
2345
  sessionId: string | undefined,
2341
2346
  body: RequestBody,
@@ -2400,7 +2405,7 @@ async function openCodexWebSocketEventStream(
2400
2405
 
2401
2406
  function createCodexHeaders(
2402
2407
  initHeaders: Record<string, string> | undefined,
2403
- accountId: string,
2408
+ accountId: string | undefined,
2404
2409
  accessToken: string,
2405
2410
  promptCacheKey?: string,
2406
2411
  transport: CodexTransport = "sse",
@@ -2409,7 +2414,11 @@ function createCodexHeaders(
2409
2414
  const headers = new Headers(initHeaders ?? {});
2410
2415
  headers.delete("x-api-key");
2411
2416
  headers.set("Authorization", `Bearer ${accessToken}`);
2412
- headers.set(OPENAI_HEADERS.ACCOUNT_ID, accountId);
2417
+ if (accountId) {
2418
+ headers.set(OPENAI_HEADERS.ACCOUNT_ID, accountId);
2419
+ } else {
2420
+ headers.delete(OPENAI_HEADERS.ACCOUNT_ID);
2421
+ }
2413
2422
  const betaHeader =
2414
2423
  transport === "websocket"
2415
2424
  ? OPENAI_HEADER_VALUES.BETA_RESPONSES_WEBSOCKETS_V2
@@ -2482,12 +2491,8 @@ function resolveCodexResponsesUrl(baseUrl: string | undefined): string {
2482
2491
  return `${normalized}/codex/responses`;
2483
2492
  }
2484
2493
 
2485
- function getAccountId(accessToken: string): string {
2486
- const accountId = getCodexAccountId(accessToken);
2487
- if (!accountId) {
2488
- throw new Error("Failed to extract accountId from token");
2489
- }
2490
- return accountId;
2494
+ function getAccountId(accessToken: string): string | undefined {
2495
+ return getCodexAccountId(accessToken);
2491
2496
  }
2492
2497
 
2493
2498
  function convertMessages(model: Model<"openai-codex-responses">, context: Context): ResponseInput {
@@ -2669,6 +2674,22 @@ function getString(value: unknown): string | undefined {
2669
2674
  return typeof value === "string" ? value : undefined;
2670
2675
  }
2671
2676
 
2677
+ function getCodexEventError(rawEvent: Record<string, unknown>): Record<string, unknown> | null {
2678
+ const response = asRecord(rawEvent.response);
2679
+ return asRecord(rawEvent.error) ?? (response ? asRecord(response.error) : null);
2680
+ }
2681
+
2682
+ function getCodexEventErrorCode(rawEvent: Record<string, unknown>): string {
2683
+ const error = getCodexEventError(rawEvent);
2684
+ return getString(error?.code) ?? getString(error?.type) ?? getString(rawEvent.code) ?? "";
2685
+ }
2686
+
2687
+ function getCodexEventErrorMessage(rawEvent: Record<string, unknown>): string {
2688
+ const response = asRecord(rawEvent.response);
2689
+ const error = getCodexEventError(rawEvent);
2690
+ return getString(error?.message) ?? getString(rawEvent.message) ?? getString(response?.message) ?? "";
2691
+ }
2692
+
2672
2693
  class CodexProviderStreamError extends Error {
2673
2694
  readonly retryable: boolean;
2674
2695
  readonly code?: string;
@@ -2682,19 +2703,17 @@ class CodexProviderStreamError extends Error {
2682
2703
  }
2683
2704
 
2684
2705
  function isRetryableCodexFailureEvent(rawEvent: Record<string, unknown>): boolean {
2685
- const response = asRecord(rawEvent.response);
2686
- const error = asRecord(rawEvent.error) ?? (response ? asRecord(response.error) : null);
2687
- const code = getString(error?.code) ?? getString(error?.type) ?? getString(rawEvent.code);
2706
+ const code = getCodexEventErrorCode(rawEvent);
2688
2707
  if (code && CODEX_RETRYABLE_EVENT_CODES.has(code.toLowerCase())) {
2689
2708
  return true;
2690
2709
  }
2691
- const message = getString(error?.message) ?? getString(rawEvent.message) ?? getString(response?.message);
2710
+ const message = getCodexEventErrorMessage(rawEvent);
2692
2711
  return !!message && CODEX_RETRYABLE_EVENT_MESSAGE.test(message);
2693
2712
  }
2694
2713
 
2695
2714
  function createCodexProviderStreamError(rawEvent: Record<string, unknown>): CodexProviderStreamError {
2696
- const code = getString(rawEvent.code) ?? "";
2697
- const message = getString(rawEvent.message) ?? "";
2715
+ const code = getCodexEventErrorCode(rawEvent);
2716
+ const message = getCodexEventErrorMessage(rawEvent);
2698
2717
  const formattedMessage =
2699
2718
  typeof rawEvent.type === "string" && rawEvent.type === "error"
2700
2719
  ? formatCodexErrorEvent(rawEvent, code, message)