@gajae-code/ai 0.8.2 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +6 -0
- package/dist/types/providers/openai-bounded-rate-limits.d.ts +3 -0
- package/package.json +2 -2
- package/src/providers/azure-openai-responses.ts +2 -1
- package/src/providers/openai-bounded-rate-limits.ts +57 -0
- package/src/providers/openai-completions-compat.ts +26 -13
- package/src/providers/openai-completions.ts +5 -1
- package/src/providers/openai-responses.ts +5 -1
- package/src/utils.ts +21 -1
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,12 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [0.9.0] - 2026-07-07
|
|
6
|
+
|
|
7
|
+
### Fixed
|
|
8
|
+
|
|
9
|
+
- Capped OpenCode Go Kimi reasoning efforts that the Go chat-completions endpoint rejects (`kimi-k2.5:minimal` → `low`, `kimi-k2.7-code:xhigh|max` → `high`) and degraded forced `tool_choice` for those models so Kimi Go sessions and title-generation turns no longer fail with generic upstream 400s.
|
|
10
|
+
|
|
5
11
|
## [0.8.2] - 2026-07-06
|
|
6
12
|
|
|
7
13
|
### Fixed
|
|
@@ -0,0 +1,3 @@
|
|
|
1
|
+
import type { FetchImpl } from "../types";
|
|
2
|
+
export declare function isOpenAIUsageExhaustionResponse(bodyText: string, retryAfterMs: number | undefined, retryDelayCapMs: number): boolean;
|
|
3
|
+
export declare function wrapOpenAIFetchForBoundedRateLimits(baseFetch: FetchImpl, maxRetryDelayMs: number | undefined): FetchImpl;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@gajae-code/ai",
|
|
4
|
-
"version": "0.
|
|
4
|
+
"version": "0.9.0",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://gajae-code.com",
|
|
7
7
|
"author": "Yeachan-Heo",
|
|
@@ -43,7 +43,7 @@
|
|
|
43
43
|
"dependencies": {
|
|
44
44
|
"@anthropic-ai/sdk": "^0.94.0",
|
|
45
45
|
"@bufbuild/protobuf": "^2.12.0",
|
|
46
|
-
"@gajae-code/utils": "0.
|
|
46
|
+
"@gajae-code/utils": "0.9.0",
|
|
47
47
|
"openai": "^6.36.0",
|
|
48
48
|
"partial-json": "^0.1.7",
|
|
49
49
|
"zod": "4.4.3"
|
|
@@ -35,6 +35,7 @@ import {
|
|
|
35
35
|
markToolChoiceIncapability,
|
|
36
36
|
resolveToolChoice,
|
|
37
37
|
} from "../utils/tool-choice-capability";
|
|
38
|
+
import { wrapOpenAIFetchForBoundedRateLimits } from "./openai-bounded-rate-limits";
|
|
38
39
|
import { normalizeOpenAIResponsesPromptCacheKey, supportsDeveloperRole } from "./openai-responses";
|
|
39
40
|
import {
|
|
40
41
|
appendResponsesToolResultMessages,
|
|
@@ -272,7 +273,7 @@ function createClient(model: Model<"azure-openai-responses">, apiKey: string, op
|
|
|
272
273
|
|
|
273
274
|
const { baseUrl, apiVersion } = resolveAzureConfig(model, options);
|
|
274
275
|
|
|
275
|
-
const baseFetch = options?.fetch ?? fetch;
|
|
276
|
+
const baseFetch = wrapOpenAIFetchForBoundedRateLimits(options?.fetch ?? fetch, options?.maxRetryDelayMs);
|
|
276
277
|
const onSseEvent = options?.onSseEvent;
|
|
277
278
|
return new AzureOpenAI({
|
|
278
279
|
apiKey,
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
import type { FetchImpl } from "../types";
|
|
2
|
+
import { getRetryAfterMsFromHeaders } from "../utils/retry-after";
|
|
3
|
+
|
|
4
|
+
const OPENAI_RETRY_DELAY_CAP_MS = 60_000;
|
|
5
|
+
|
|
6
|
+
// Mirror of `wrapAnthropicFetchForBoundedRateLimits`: OpenAI-compatible providers
|
|
7
|
+
// (e.g. opencode-go) return HTTP 429 for *permanent* usage/quota exhaustion — a
|
|
8
|
+
// monthly-cap reset that can be days away. The OpenAI SDK treats 429 as transient
|
|
9
|
+
// and retries up to `maxRetries`, honoring an out-of-range `Retry-After`; the
|
|
10
|
+
// `create()` call then hangs before the error can surface to the agent loop, so
|
|
11
|
+
// no assistant error is produced and the session-level retry/fallback never runs.
|
|
12
|
+
// Detect exhaustion and set `x-should-retry: false` so the SDK gives up at once
|
|
13
|
+
// and the session retry layer applies its own fail-fast (retry-after > maxDelayMs).
|
|
14
|
+
//
|
|
15
|
+
// Shared by every adapter that drives a raw OpenAI SDK client — openai-completions,
|
|
16
|
+
// openai-responses, and azure-openai-responses. Adapters that route through
|
|
17
|
+
// `fetchWithRetry` (codex, bedrock, ollama, gemini-cli) already bound 429 retries
|
|
18
|
+
// themselves and do not need this wrapper.
|
|
19
|
+
export function isOpenAIUsageExhaustionResponse(
|
|
20
|
+
bodyText: string,
|
|
21
|
+
retryAfterMs: number | undefined,
|
|
22
|
+
retryDelayCapMs: number,
|
|
23
|
+
): boolean {
|
|
24
|
+
if (retryAfterMs !== undefined && retryAfterMs > retryDelayCapMs) return true;
|
|
25
|
+
return /monthly usage limit|usage limit reached|usage_limit_reached|out_of_credits|insufficient_quota|quota[ _]?exceeded/i.test(
|
|
26
|
+
bodyText,
|
|
27
|
+
);
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export function wrapOpenAIFetchForBoundedRateLimits(
|
|
31
|
+
baseFetch: FetchImpl,
|
|
32
|
+
maxRetryDelayMs: number | undefined,
|
|
33
|
+
): FetchImpl {
|
|
34
|
+
const retryDelayCapMs = maxRetryDelayMs ?? OPENAI_RETRY_DELAY_CAP_MS;
|
|
35
|
+
return Object.assign(
|
|
36
|
+
async (input: string | URL | Request, init?: RequestInit): Promise<Response> => {
|
|
37
|
+
const response = await baseFetch(input, init);
|
|
38
|
+
if (response.status !== 429 || retryDelayCapMs === 0) return response;
|
|
39
|
+
|
|
40
|
+
const headers = new Headers(response.headers);
|
|
41
|
+
const retryAfterMs = getRetryAfterMsFromHeaders(headers);
|
|
42
|
+
const bodyText = await response
|
|
43
|
+
.clone()
|
|
44
|
+
.text()
|
|
45
|
+
.catch(() => "");
|
|
46
|
+
if (!isOpenAIUsageExhaustionResponse(bodyText, retryAfterMs, retryDelayCapMs)) return response;
|
|
47
|
+
|
|
48
|
+
headers.set("x-should-retry", "false");
|
|
49
|
+
return new Response(bodyText, {
|
|
50
|
+
status: response.status,
|
|
51
|
+
statusText: response.statusText,
|
|
52
|
+
headers,
|
|
53
|
+
});
|
|
54
|
+
},
|
|
55
|
+
baseFetch.preconnect ? { preconnect: baseFetch.preconnect } : {},
|
|
56
|
+
);
|
|
57
|
+
}
|
|
@@ -104,6 +104,10 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
|
|
104
104
|
baseUrl.includes("opencode.ai");
|
|
105
105
|
const isOpenCodeProvider = provider === "opencode-go" || provider === "opencode-zen";
|
|
106
106
|
const isOpenCodeGoReasoning = provider === "opencode-go" && Boolean(model.reasoning);
|
|
107
|
+
const isOpenCodeGoKimiReasoning = provider === "opencode-go" && isKimiModel && Boolean(model.reasoning);
|
|
108
|
+
const isOpenCodeGoKimi25Reasoning = isOpenCodeGoKimiReasoning && model.id === "kimi-k2.5";
|
|
109
|
+
const isOpenCodeGoKimi27CodeReasoning = isOpenCodeGoKimiReasoning && model.id === "kimi-k2.7-code";
|
|
110
|
+
const needsOpenCodeGoKimiEffortMap = isOpenCodeGoKimi25Reasoning || isOpenCodeGoKimi27CodeReasoning;
|
|
107
111
|
|
|
108
112
|
const useMaxTokens =
|
|
109
113
|
provider === "mistral" ||
|
|
@@ -170,22 +174,31 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
|
|
170
174
|
xhigh: "default",
|
|
171
175
|
max: "default",
|
|
172
176
|
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
|
|
173
|
-
:
|
|
177
|
+
: needsOpenCodeGoKimiEffortMap
|
|
174
178
|
? ({
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
179
|
+
// Live Go probes (2026-07-06) showed model-specific effort gaps:
|
|
180
|
+
// kimi-k2.5 rejects "minimal", while kimi-k2.7-code rejects
|
|
181
|
+
// OpenAI-style "xhigh" and "max"; all other Kimi efforts tested
|
|
182
|
+
// successfully and should pass through unchanged.
|
|
183
|
+
...(isOpenCodeGoKimi25Reasoning ? { minimal: "low" } : {}),
|
|
184
|
+
...(isOpenCodeGoKimi27CodeReasoning ? { xhigh: "high", max: "high" } : {}),
|
|
181
185
|
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
|
|
182
|
-
:
|
|
186
|
+
: isDeepseekFamily && model.reasoning
|
|
183
187
|
? ({
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
188
|
+
minimal: "high",
|
|
189
|
+
low: "high",
|
|
190
|
+
medium: "high",
|
|
191
|
+
high: "high",
|
|
192
|
+
xhigh: "max",
|
|
193
|
+
max: "max",
|
|
187
194
|
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
|
|
188
|
-
:
|
|
195
|
+
: isFireworks
|
|
196
|
+
? ({
|
|
197
|
+
// Fireworks' OpenAI-compatible endpoint rejects OpenAI's
|
|
198
|
+
// `minimal` literal but accepts `none` for the lowest setting.
|
|
199
|
+
minimal: "none",
|
|
200
|
+
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
|
|
201
|
+
: {};
|
|
189
202
|
|
|
190
203
|
return {
|
|
191
204
|
supportsStore: !isNonStandard,
|
|
@@ -198,7 +211,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
|
|
198
211
|
disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel || isOpenCodeGoReasoning,
|
|
199
212
|
disableReasoningOnToolChoice: isDeepseekFamily && Boolean(model.reasoning) && !isOpenRouter,
|
|
200
213
|
supportsToolChoice: !isDirectDeepseekReasoning,
|
|
201
|
-
supportsForcedToolChoice:
|
|
214
|
+
supportsForcedToolChoice: !isOpenCodeGoKimiReasoning,
|
|
202
215
|
maxTokensField: useMaxTokens ? "max_tokens" : "max_completion_tokens",
|
|
203
216
|
requiresToolResultName: isMistral,
|
|
204
217
|
requiresAssistantAfterToolResult: false,
|
|
@@ -72,6 +72,7 @@ import {
|
|
|
72
72
|
hasCopilotVisionInput,
|
|
73
73
|
resolveGitHubCopilotBaseUrl,
|
|
74
74
|
} from "./github-copilot-headers";
|
|
75
|
+
import { wrapOpenAIFetchForBoundedRateLimits } from "./openai-bounded-rate-limits";
|
|
75
76
|
import { detectOpenAICompat, type ResolvedOpenAICompat, resolveOpenAICompat } from "./openai-completions-compat";
|
|
76
77
|
import {
|
|
77
78
|
applyOpenAIRequestTransformBody,
|
|
@@ -454,6 +455,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
454
455
|
options?.authCredentialType,
|
|
455
456
|
options?.requestMaxRetries,
|
|
456
457
|
options?.sessionId,
|
|
458
|
+
options?.maxRetryDelayMs,
|
|
457
459
|
);
|
|
458
460
|
const premiumRequestsTotal = copilotPremiumRequests;
|
|
459
461
|
getCapturedErrorResponse = captureErrorResponse;
|
|
@@ -956,6 +958,7 @@ async function createClient(
|
|
|
956
958
|
authCredentialType?: OpenAICompletionsOptions["authCredentialType"],
|
|
957
959
|
requestMaxRetries?: number,
|
|
958
960
|
sessionId?: string,
|
|
961
|
+
maxRetryDelayMs?: number,
|
|
959
962
|
): Promise<{
|
|
960
963
|
client: OpenAI;
|
|
961
964
|
copilotPremiumRequests: number | undefined;
|
|
@@ -1063,8 +1066,9 @@ async function createClient(
|
|
|
1063
1066
|
},
|
|
1064
1067
|
baseFetch.preconnect ? { preconnect: baseFetch.preconnect } : {},
|
|
1065
1068
|
);
|
|
1069
|
+
const boundedFetch = wrapOpenAIFetchForBoundedRateLimits(wrappedFetch, maxRetryDelayMs);
|
|
1066
1070
|
const transformedFetch = wrapFetchForOpenAIRequestTransform(
|
|
1067
|
-
|
|
1071
|
+
boundedFetch,
|
|
1068
1072
|
model.requestTransform,
|
|
1069
1073
|
`Gajae-Code/${packageJson.version}`,
|
|
1070
1074
|
);
|
|
@@ -71,6 +71,7 @@ import {
|
|
|
71
71
|
resolveGitHubCopilotBaseUrl,
|
|
72
72
|
} from "./github-copilot-headers";
|
|
73
73
|
import { compactGrammarDefinition } from "./grammar";
|
|
74
|
+
import { wrapOpenAIFetchForBoundedRateLimits } from "./openai-bounded-rate-limits";
|
|
74
75
|
import {
|
|
75
76
|
applyOpenAIRequestTransformBody,
|
|
76
77
|
applyOpenAIRequestTransformHeaders,
|
|
@@ -274,6 +275,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
|
|
274
275
|
options?.fetch,
|
|
275
276
|
options?.authCredentialType,
|
|
276
277
|
options?.requestMaxRetries,
|
|
278
|
+
options?.maxRetryDelayMs,
|
|
277
279
|
);
|
|
278
280
|
const premiumRequestsTotal = copilotPremiumRequests;
|
|
279
281
|
const providerSessionState = getOpenAIResponsesProviderSessionState(model, options?.providerSessionState);
|
|
@@ -400,6 +402,7 @@ function createClient(
|
|
|
400
402
|
fetchOverride?: FetchImpl,
|
|
401
403
|
authCredentialType?: OpenAIResponsesOptions["authCredentialType"],
|
|
402
404
|
requestMaxRetries?: number,
|
|
405
|
+
maxRetryDelayMs?: number,
|
|
403
406
|
): {
|
|
404
407
|
client: OpenAI;
|
|
405
408
|
copilotPremiumRequests: number | undefined;
|
|
@@ -446,8 +449,9 @@ function createClient(
|
|
|
446
449
|
headers["x-client-request-id"] ??= sessionId;
|
|
447
450
|
}
|
|
448
451
|
const baseFetch = fetchOverride ?? fetch;
|
|
452
|
+
const boundedFetch = wrapOpenAIFetchForBoundedRateLimits(baseFetch, maxRetryDelayMs);
|
|
449
453
|
const transformedFetch = wrapFetchForOpenAIRequestTransform(
|
|
450
|
-
|
|
454
|
+
boundedFetch,
|
|
451
455
|
model.requestTransform,
|
|
452
456
|
`Gajae-Code/${packageJson.version}`,
|
|
453
457
|
);
|
package/src/utils.ts
CHANGED
|
@@ -215,7 +215,11 @@ function sanitizeOpenAIResponsesHistoryItemForReplay(
|
|
|
215
215
|
// providerPayload stores raw output items; replay strips fields that are output-only.
|
|
216
216
|
const { id: _id, ...itemWithoutId } = item;
|
|
217
217
|
const sanitizedItem =
|
|
218
|
-
item.type === "computer_call"
|
|
218
|
+
item.type === "computer_call"
|
|
219
|
+
? sanitizeComputerCallForResponsesInput(itemWithoutId)
|
|
220
|
+
: item.type === "image_generation_call"
|
|
221
|
+
? sanitizeImageGenerationCallForResponsesInput(itemWithoutId)
|
|
222
|
+
: itemWithoutId;
|
|
219
223
|
if (typeof item.call_id === "string") {
|
|
220
224
|
sanitizedItem.call_id = normalizeReplayedResponsesHistoryCallId(item.call_id, normalizedCallIds);
|
|
221
225
|
}
|
|
@@ -231,6 +235,22 @@ function sanitizeComputerCallForResponsesInput(item: Record<string, unknown>): R
|
|
|
231
235
|
return inputSafeItem;
|
|
232
236
|
}
|
|
233
237
|
|
|
238
|
+
function sanitizeImageGenerationCallForResponsesInput(item: Record<string, unknown>): Record<string, unknown> {
|
|
239
|
+
// Image generation output items include request-time knobs that are not part of
|
|
240
|
+
// the Responses input replay schema. Replaying them verbatim makes OpenAI-compatible
|
|
241
|
+
// endpoints reject the next turn, e.g. `Unknown parameter: input[n].action`.
|
|
242
|
+
const {
|
|
243
|
+
action: _action,
|
|
244
|
+
background: _background,
|
|
245
|
+
output_format: _outputFormat,
|
|
246
|
+
quality: _quality,
|
|
247
|
+
revised_prompt: _revisedPrompt,
|
|
248
|
+
size: _size,
|
|
249
|
+
...inputSafeItem
|
|
250
|
+
} = item;
|
|
251
|
+
return inputSafeItem;
|
|
252
|
+
}
|
|
253
|
+
|
|
234
254
|
function normalizeReplayedResponsesHistoryCallId(value: string, normalizedValues: Map<string, string>): string {
|
|
235
255
|
const normalized = normalizedValues.get(value);
|
|
236
256
|
if (normalized) return normalized;
|