@gajae-code/ai 0.8.2 → 0.9.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,20 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.9.1] - 2026-07-08
6
+
7
+ ### Fixed
8
+
9
+ - Unified the Cursor client version used across provider requests and discovery.
10
+ - Detected ZAI weekly limit exhaustion as a structured rate-limit condition.
11
+ - Pointed Sakana Fugu OAuth/login guidance at the Sakana platform console and documented the `fish_` key prefix expectation.
12
+
13
+ ## [0.9.0] - 2026-07-07
14
+
15
+ ### Fixed
16
+
17
+ - Capped OpenCode Go Kimi reasoning efforts that the Go chat-completions endpoint rejects (`kimi-k2.5:minimal` → `low`, `kimi-k2.7-code:xhigh|max` → `high`) and degraded forced `tool_choice` for those models so Kimi Go sessions and title-generation turns no longer fail with generic upstream 400s.
18
+
5
19
  ## [0.8.2] - 2026-07-06
6
20
 
7
21
  ### Fixed
@@ -0,0 +1,10 @@
1
+ /**
2
+ * The Cursor client version reported to api2.cursor.sh.
3
+ *
4
+ * Every call against the Cursor backend must send the same
5
+ * x-cursor-client-version: the backend gates features and minimum versions on
6
+ * it, so a drift between the agent Run path and model discovery makes one of
7
+ * them fail while the other keeps working. Keep this as the single source of
8
+ * truth for the header value.
9
+ */
10
+ export declare const CURSOR_CLIENT_VERSION = "cli-2026.02.13-41ac335";
@@ -1,7 +1,8 @@
1
1
  import { type JsonValue } from "@bufbuild/protobuf";
2
2
  import type { CursorExecHandlerResult, CursorExecHandlers, CursorToolResultHandler, Message, StreamFunction, StreamOptions, ToolResultMessage } from "../types";
3
+ import { CURSOR_CLIENT_VERSION } from "./cursor/client-version";
3
4
  export declare const CURSOR_API_URL = "https://api2.cursor.sh";
4
- export declare const CURSOR_CLIENT_VERSION = "cli-2026.01.09-231024f";
5
+ export { CURSOR_CLIENT_VERSION };
5
6
  /** Drop all cached state + blob bytes for a conversation (F15 bound + session-teardown hook). */
6
7
  export declare function disposeCursorConversation(conversationId: string): void;
7
8
  export interface CursorOptions extends StreamOptions {
@@ -0,0 +1,3 @@
1
+ import type { FetchImpl } from "../types";
2
+ export declare function isOpenAIUsageExhaustionResponse(bodyText: string, retryAfterMs: number | undefined, retryDelayCapMs: number): boolean;
3
+ export declare function wrapOpenAIFetchForBoundedRateLimits(baseFetch: FetchImpl, maxRetryDelayMs: number | undefined): FetchImpl;
package/package.json CHANGED
@@ -1,13 +1,10 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/ai",
4
- "version": "0.8.2",
4
+ "version": "0.9.1",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://gajae-code.com",
7
- "author": "Yeachan-Heo",
8
- "contributors": [
9
- "Mario Zechner"
10
- ],
7
+ "author": "Yeachan-Heo and Gajae Code Contributors",
11
8
  "license": "MIT",
12
9
  "repository": {
13
10
  "type": "git",
@@ -43,7 +40,7 @@
43
40
  "dependencies": {
44
41
  "@anthropic-ai/sdk": "^0.94.0",
45
42
  "@bufbuild/protobuf": "^2.12.0",
46
- "@gajae-code/utils": "0.8.2",
43
+ "@gajae-code/utils": "0.9.1",
47
44
  "openai": "^6.36.0",
48
45
  "partial-json": "^0.1.7",
49
46
  "zod": "4.4.3"
@@ -35,6 +35,7 @@ import {
35
35
  markToolChoiceIncapability,
36
36
  resolveToolChoice,
37
37
  } from "../utils/tool-choice-capability";
38
+ import { wrapOpenAIFetchForBoundedRateLimits } from "./openai-bounded-rate-limits";
38
39
  import { normalizeOpenAIResponsesPromptCacheKey, supportsDeveloperRole } from "./openai-responses";
39
40
  import {
40
41
  appendResponsesToolResultMessages,
@@ -272,7 +273,7 @@ function createClient(model: Model<"azure-openai-responses">, apiKey: string, op
272
273
 
273
274
  const { baseUrl, apiVersion } = resolveAzureConfig(model, options);
274
275
 
275
- const baseFetch = options?.fetch ?? fetch;
276
+ const baseFetch = wrapOpenAIFetchForBoundedRateLimits(options?.fetch ?? fetch, options?.maxRetryDelayMs);
276
277
  const onSseEvent = options?.onSseEvent;
277
278
  return new AzureOpenAI({
278
279
  apiKey,
@@ -0,0 +1,10 @@
1
+ /**
2
+ * The Cursor client version reported to api2.cursor.sh.
3
+ *
4
+ * Every call against the Cursor backend must send the same
5
+ * x-cursor-client-version: the backend gates features and minimum versions on
6
+ * it, so a drift between the agent Run path and model discovery makes one of
7
+ * them fail while the other keeps working. Keep this as the single source of
8
+ * truth for the header value.
9
+ */
10
+ export const CURSOR_CLIENT_VERSION = "cli-2026.02.13-41ac335";
@@ -31,6 +31,7 @@ import { parseStreamingJson } from "../utils/json-parse";
31
31
  import { formatErrorMessageWithRetryAfter } from "../utils/retry-after";
32
32
  import { flattenToolRootCombinators, toolWireSchema } from "../utils/schema";
33
33
  import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
34
+ import { CURSOR_CLIENT_VERSION } from "./cursor/client-version";
34
35
  import type { McpToolDefinition } from "./cursor/gen/agent_pb";
35
36
  import {
36
37
  AgentClientMessageSchema,
@@ -131,7 +132,7 @@ import {
131
132
  } from "./cursor/gen/agent_pb";
132
133
 
133
134
  export const CURSOR_API_URL = "https://api2.cursor.sh";
134
- export const CURSOR_CLIENT_VERSION = "cli-2026.01.09-231024f";
135
+ export { CURSOR_CLIENT_VERSION };
135
136
 
136
137
  const conversationStateCache = new Map<string, ConversationStateStructure>();
137
138
  const conversationBlobStores = new Map<string, Map<string, Uint8Array>>();
@@ -0,0 +1,57 @@
1
+ import type { FetchImpl } from "../types";
2
+ import { getRetryAfterMsFromHeaders } from "../utils/retry-after";
3
+
4
+ const OPENAI_RETRY_DELAY_CAP_MS = 60_000;
5
+
6
+ // Mirror of `wrapAnthropicFetchForBoundedRateLimits`: OpenAI-compatible providers
7
+ // (e.g. opencode-go) return HTTP 429 for *permanent* usage/quota exhaustion — a
8
+ // monthly-cap reset that can be days away. The OpenAI SDK treats 429 as transient
9
+ // and retries up to `maxRetries`, honoring an out-of-range `Retry-After`; the
10
+ // `create()` call then hangs before the error can surface to the agent loop, so
11
+ // no assistant error is produced and the session-level retry/fallback never runs.
12
+ // Detect exhaustion and set `x-should-retry: false` so the SDK gives up at once
13
+ // and the session retry layer applies its own fail-fast (retry-after > maxDelayMs).
14
+ //
15
+ // Shared by every adapter that drives a raw OpenAI SDK client — openai-completions,
16
+ // openai-responses, and azure-openai-responses. Adapters that route through
17
+ // `fetchWithRetry` (codex, bedrock, ollama, gemini-cli) already bound 429 retries
18
+ // themselves and do not need this wrapper.
19
+ export function isOpenAIUsageExhaustionResponse(
20
+ bodyText: string,
21
+ retryAfterMs: number | undefined,
22
+ retryDelayCapMs: number,
23
+ ): boolean {
24
+ if (retryAfterMs !== undefined && retryAfterMs > retryDelayCapMs) return true;
25
+ return /monthly usage limit|usage limit reached|usage_limit_reached|out_of_credits|insufficient_quota|quota[ _]?exceeded/i.test(
26
+ bodyText,
27
+ );
28
+ }
29
+
30
+ export function wrapOpenAIFetchForBoundedRateLimits(
31
+ baseFetch: FetchImpl,
32
+ maxRetryDelayMs: number | undefined,
33
+ ): FetchImpl {
34
+ const retryDelayCapMs = maxRetryDelayMs ?? OPENAI_RETRY_DELAY_CAP_MS;
35
+ return Object.assign(
36
+ async (input: string | URL | Request, init?: RequestInit): Promise<Response> => {
37
+ const response = await baseFetch(input, init);
38
+ if (response.status !== 429 || retryDelayCapMs === 0) return response;
39
+
40
+ const headers = new Headers(response.headers);
41
+ const retryAfterMs = getRetryAfterMsFromHeaders(headers);
42
+ const bodyText = await response
43
+ .clone()
44
+ .text()
45
+ .catch(() => "");
46
+ if (!isOpenAIUsageExhaustionResponse(bodyText, retryAfterMs, retryDelayCapMs)) return response;
47
+
48
+ headers.set("x-should-retry", "false");
49
+ return new Response(bodyText, {
50
+ status: response.status,
51
+ statusText: response.statusText,
52
+ headers,
53
+ });
54
+ },
55
+ baseFetch.preconnect ? { preconnect: baseFetch.preconnect } : {},
56
+ );
57
+ }
@@ -104,6 +104,10 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
104
104
  baseUrl.includes("opencode.ai");
105
105
  const isOpenCodeProvider = provider === "opencode-go" || provider === "opencode-zen";
106
106
  const isOpenCodeGoReasoning = provider === "opencode-go" && Boolean(model.reasoning);
107
+ const isOpenCodeGoKimiReasoning = provider === "opencode-go" && isKimiModel && Boolean(model.reasoning);
108
+ const isOpenCodeGoKimi25Reasoning = isOpenCodeGoKimiReasoning && model.id === "kimi-k2.5";
109
+ const isOpenCodeGoKimi27CodeReasoning = isOpenCodeGoKimiReasoning && model.id === "kimi-k2.7-code";
110
+ const needsOpenCodeGoKimiEffortMap = isOpenCodeGoKimi25Reasoning || isOpenCodeGoKimi27CodeReasoning;
107
111
 
108
112
  const useMaxTokens =
109
113
  provider === "mistral" ||
@@ -170,22 +174,31 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
170
174
  xhigh: "default",
171
175
  max: "default",
172
176
  } satisfies Partial<Record<OpenAIReasoningEffort, string>>)
173
- : isDeepseekFamily && model.reasoning
177
+ : needsOpenCodeGoKimiEffortMap
174
178
  ? ({
175
- minimal: "high",
176
- low: "high",
177
- medium: "high",
178
- high: "high",
179
- xhigh: "max",
180
- max: "max",
179
+ // Live Go probes (2026-07-06) showed model-specific effort gaps:
180
+ // kimi-k2.5 rejects "minimal", while kimi-k2.7-code rejects
181
+ // OpenAI-style "xhigh" and "max"; all other Kimi efforts tested
182
+ // successfully and should pass through unchanged.
183
+ ...(isOpenCodeGoKimi25Reasoning ? { minimal: "low" } : {}),
184
+ ...(isOpenCodeGoKimi27CodeReasoning ? { xhigh: "high", max: "high" } : {}),
181
185
  } satisfies Partial<Record<OpenAIReasoningEffort, string>>)
182
- : isFireworks
186
+ : isDeepseekFamily && model.reasoning
183
187
  ? ({
184
- // Fireworks' OpenAI-compatible endpoint rejects OpenAI's
185
- // `minimal` literal but accepts `none` for the lowest setting.
186
- minimal: "none",
188
+ minimal: "high",
189
+ low: "high",
190
+ medium: "high",
191
+ high: "high",
192
+ xhigh: "max",
193
+ max: "max",
187
194
  } satisfies Partial<Record<OpenAIReasoningEffort, string>>)
188
- : {};
195
+ : isFireworks
196
+ ? ({
197
+ // Fireworks' OpenAI-compatible endpoint rejects OpenAI's
198
+ // `minimal` literal but accepts `none` for the lowest setting.
199
+ minimal: "none",
200
+ } satisfies Partial<Record<OpenAIReasoningEffort, string>>)
201
+ : {};
189
202
 
190
203
  return {
191
204
  supportsStore: !isNonStandard,
@@ -198,7 +211,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
198
211
  disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel || isOpenCodeGoReasoning,
199
212
  disableReasoningOnToolChoice: isDeepseekFamily && Boolean(model.reasoning) && !isOpenRouter,
200
213
  supportsToolChoice: !isDirectDeepseekReasoning,
201
- supportsForcedToolChoice: true,
214
+ supportsForcedToolChoice: !isOpenCodeGoKimiReasoning,
202
215
  maxTokensField: useMaxTokens ? "max_tokens" : "max_completion_tokens",
203
216
  requiresToolResultName: isMistral,
204
217
  requiresAssistantAfterToolResult: false,
@@ -72,6 +72,7 @@ import {
72
72
  hasCopilotVisionInput,
73
73
  resolveGitHubCopilotBaseUrl,
74
74
  } from "./github-copilot-headers";
75
+ import { wrapOpenAIFetchForBoundedRateLimits } from "./openai-bounded-rate-limits";
75
76
  import { detectOpenAICompat, type ResolvedOpenAICompat, resolveOpenAICompat } from "./openai-completions-compat";
76
77
  import {
77
78
  applyOpenAIRequestTransformBody,
@@ -454,6 +455,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
454
455
  options?.authCredentialType,
455
456
  options?.requestMaxRetries,
456
457
  options?.sessionId,
458
+ options?.maxRetryDelayMs,
457
459
  );
458
460
  const premiumRequestsTotal = copilotPremiumRequests;
459
461
  getCapturedErrorResponse = captureErrorResponse;
@@ -956,6 +958,7 @@ async function createClient(
956
958
  authCredentialType?: OpenAICompletionsOptions["authCredentialType"],
957
959
  requestMaxRetries?: number,
958
960
  sessionId?: string,
961
+ maxRetryDelayMs?: number,
959
962
  ): Promise<{
960
963
  client: OpenAI;
961
964
  copilotPremiumRequests: number | undefined;
@@ -1063,8 +1066,9 @@ async function createClient(
1063
1066
  },
1064
1067
  baseFetch.preconnect ? { preconnect: baseFetch.preconnect } : {},
1065
1068
  );
1069
+ const boundedFetch = wrapOpenAIFetchForBoundedRateLimits(wrappedFetch, maxRetryDelayMs);
1066
1070
  const transformedFetch = wrapFetchForOpenAIRequestTransform(
1067
- wrappedFetch,
1071
+ boundedFetch,
1068
1072
  model.requestTransform,
1069
1073
  `Gajae-Code/${packageJson.version}`,
1070
1074
  );
@@ -71,6 +71,7 @@ import {
71
71
  resolveGitHubCopilotBaseUrl,
72
72
  } from "./github-copilot-headers";
73
73
  import { compactGrammarDefinition } from "./grammar";
74
+ import { wrapOpenAIFetchForBoundedRateLimits } from "./openai-bounded-rate-limits";
74
75
  import {
75
76
  applyOpenAIRequestTransformBody,
76
77
  applyOpenAIRequestTransformHeaders,
@@ -274,6 +275,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
274
275
  options?.fetch,
275
276
  options?.authCredentialType,
276
277
  options?.requestMaxRetries,
278
+ options?.maxRetryDelayMs,
277
279
  );
278
280
  const premiumRequestsTotal = copilotPremiumRequests;
279
281
  const providerSessionState = getOpenAIResponsesProviderSessionState(model, options?.providerSessionState);
@@ -400,6 +402,7 @@ function createClient(
400
402
  fetchOverride?: FetchImpl,
401
403
  authCredentialType?: OpenAIResponsesOptions["authCredentialType"],
402
404
  requestMaxRetries?: number,
405
+ maxRetryDelayMs?: number,
403
406
  ): {
404
407
  client: OpenAI;
405
408
  copilotPremiumRequests: number | undefined;
@@ -446,8 +449,9 @@ function createClient(
446
449
  headers["x-client-request-id"] ??= sessionId;
447
450
  }
448
451
  const baseFetch = fetchOverride ?? fetch;
452
+ const boundedFetch = wrapOpenAIFetchForBoundedRateLimits(baseFetch, maxRetryDelayMs);
449
453
  const transformedFetch = wrapFetchForOpenAIRequestTransform(
450
- baseFetch,
454
+ boundedFetch,
451
455
  model.requestTransform,
452
456
  `Gajae-Code/${packageJson.version}`,
453
457
  );
@@ -13,7 +13,7 @@ export type RateLimitReason =
13
13
  const QUOTA_EXHAUSTED_BACKOFF_MS = 30 * 60 * 1000; // 30 min
14
14
  const RATE_LIMIT_EXCEEDED_BACKOFF_MS = 30 * 1000; // 30s
15
15
  const MODEL_CAPACITY_BASE_MS = 45 * 1000; // 45s base
16
- const MODEL_CAPACITY_JITTER_MS = 30 * 1000; // ±15s
16
+ const MODEL_CAPACITY_JITTER_MS = 30 * 1000; // uniform +0–30s above base → 45–75s total
17
17
  const SERVER_ERROR_BACKOFF_MS = 20 * 1000; // 20s
18
18
 
19
19
  /**
@@ -93,8 +93,10 @@ export function calculateRateLimitBackoffMs(reason: RateLimitReason): number {
93
93
  }
94
94
 
95
95
  /** Detect usage/quota limit errors in error messages (persistent, requires credential switch). */
96
+ // ZAI reports durable token exhaustion as "[1310][Weekly/Monthly Limit Exhausted...]".
97
+ // Keep this explicit so generic "rate limit exhausted, retry..." throttles remain retryable.
96
98
  const USAGE_LIMIT_PATTERN =
97
- /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|model.?limit|model_limit_reached|message.?limit|message_limit_reached|limit for this model|quota.?exceeded|out_of_credits|request would exceed your account.?s rate limit|resource has been exhausted[^\n]*(?:quota|limit)/i;
99
+ /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|model.?limit|model_limit_reached|message.?limit|message_limit_reached|limit for this model|weekly\/monthly\s+limit\s+exhausted|quota.?exceeded|out_of_credits|request would exceed your account.?s rate limit|resource has been exhausted[^\n]*(?:quota|limit)/i;
98
100
  export function isUsageLimitError(errorMessage: string): boolean {
99
101
  return USAGE_LIMIT_PATTERN.test(errorMessage);
100
102
  }
@@ -2,11 +2,11 @@ import * as http2 from "node:http2";
2
2
  import { create, fromBinary, toBinary } from "@bufbuild/protobuf";
3
3
  import * as z from "zod/v4";
4
4
  import { getBundledModels } from "../../models";
5
+ import { CURSOR_CLIENT_VERSION } from "../../providers/cursor/client-version";
5
6
  import { GetUsableModelsRequestSchema, GetUsableModelsResponseSchema } from "../../providers/cursor/gen/agent_pb";
6
7
  import type { Model } from "../../types";
7
8
 
8
9
  const CURSOR_DEFAULT_BASE_URL = "https://api2.cursor.sh";
9
- const CURSOR_DEFAULT_CLIENT_VERSION = "cli-2026.02.13-41ac335";
10
10
  const CURSOR_GET_USABLE_MODELS_PATH = "/agent.v1.AgentService/GetUsableModels";
11
11
 
12
12
  const DEFAULT_CONTEXT_WINDOW = 200_000;
@@ -91,7 +91,7 @@ function buildRequestHeaders(options: CursorModelDiscoveryOptions): Record<strin
91
91
  te: "trailers",
92
92
  authorization: `Bearer ${options.apiKey}`,
93
93
  "x-ghost-mode": "true",
94
- "x-cursor-client-version": options.clientVersion ?? CURSOR_DEFAULT_CLIENT_VERSION,
94
+ "x-cursor-client-version": options.clientVersion ?? CURSOR_CLIENT_VERSION,
95
95
  "x-cursor-client-type": "cli",
96
96
  };
97
97
  }
@@ -3,10 +3,10 @@ import { createApiKeyLogin } from "./api-key-login";
3
3
 
4
4
  export const loginFugu = createApiKeyLogin({
5
5
  providerLabel: "Sakana Fugu",
6
- authUrl: "https://fugu.sakana.ai/",
6
+ authUrl: "https://console.sakana.ai/api-keys",
7
7
  instructions: "Create or copy your Sakana Fugu API key",
8
8
  promptMessage: "Paste your Sakana Fugu API key",
9
- placeholder: "fugu_...",
9
+ placeholder: "fish_...",
10
10
  validation: {
11
11
  kind: "models-endpoint",
12
12
  provider: "Sakana Fugu",
package/src/utils.ts CHANGED
@@ -215,7 +215,11 @@ function sanitizeOpenAIResponsesHistoryItemForReplay(
215
215
  // providerPayload stores raw output items; replay strips fields that are output-only.
216
216
  const { id: _id, ...itemWithoutId } = item;
217
217
  const sanitizedItem =
218
- item.type === "computer_call" ? sanitizeComputerCallForResponsesInput(itemWithoutId) : itemWithoutId;
218
+ item.type === "computer_call"
219
+ ? sanitizeComputerCallForResponsesInput(itemWithoutId)
220
+ : item.type === "image_generation_call"
221
+ ? sanitizeImageGenerationCallForResponsesInput(itemWithoutId)
222
+ : itemWithoutId;
219
223
  if (typeof item.call_id === "string") {
220
224
  sanitizedItem.call_id = normalizeReplayedResponsesHistoryCallId(item.call_id, normalizedCallIds);
221
225
  }
@@ -231,6 +235,22 @@ function sanitizeComputerCallForResponsesInput(item: Record<string, unknown>): R
231
235
  return inputSafeItem;
232
236
  }
233
237
 
238
+ function sanitizeImageGenerationCallForResponsesInput(item: Record<string, unknown>): Record<string, unknown> {
239
+ // Image generation output items include request-time knobs that are not part of
240
+ // the Responses input replay schema. Replaying them verbatim makes OpenAI-compatible
241
+ // endpoints reject the next turn, e.g. `Unknown parameter: input[n].action`.
242
+ const {
243
+ action: _action,
244
+ background: _background,
245
+ output_format: _outputFormat,
246
+ quality: _quality,
247
+ revised_prompt: _revisedPrompt,
248
+ size: _size,
249
+ ...inputSafeItem
250
+ } = item;
251
+ return inputSafeItem;
252
+ }
253
+
234
254
  function normalizeReplayedResponsesHistoryCallId(value: string, normalizedValues: Map<string, string>): string {
235
255
  const normalized = normalizedValues.get(value);
236
256
  if (normalized) return normalized;