@sayknow-cli/ai 0.3.7 → 0.3.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/types/auth-broker/remote-store.d.ts +1 -0
- package/dist/types/auth-broker/wire-schemas.d.ts +1 -0
- package/dist/types/auth-storage.d.ts +3 -1
- package/dist/types/provider-models/openai-compat.d.ts +5 -0
- package/dist/types/providers/cursor/client-version.d.ts +10 -0
- package/dist/types/providers/cursor.d.ts +2 -1
- package/dist/types/providers/openai-bounded-rate-limits.d.ts +3 -0
- package/dist/types/types.d.ts +4 -4
- package/dist/types/utils/oauth/types.d.ts +1 -1
- package/package.json +3 -3
- package/src/auth-broker/remote-store.ts +17 -2
- package/src/auth-broker/wire-schemas.ts +1 -0
- package/src/auth-storage.ts +80 -7
- package/src/provider-models/descriptors.ts +7 -0
- package/src/provider-models/openai-compat.ts +17 -0
- package/src/providers/anthropic.ts +0 -2
- package/src/providers/azure-openai-responses.ts +2 -1
- package/src/providers/cursor/client-version.ts +10 -0
- package/src/providers/cursor.ts +2 -1
- package/src/providers/google-gemini-headers.ts +1 -1
- package/src/providers/openai-bounded-rate-limits.ts +57 -0
- package/src/providers/openai-completions-compat.ts +26 -13
- package/src/providers/openai-completions.ts +3 -3
- package/src/providers/openai-responses.ts +5 -1
- package/src/rate-limit-utils.ts +4 -2
- package/src/stream.ts +1 -0
- package/src/types.ts +12 -6
- package/src/utils/discovery/cursor.ts +2 -2
- package/src/utils/oauth/fugu.ts +2 -2
- package/src/utils/oauth/index.ts +6 -0
- package/src/utils/oauth/types.ts +1 -0
- package/src/utils.ts +21 -1
|
@@ -70,6 +70,7 @@ export declare class RemoteAuthCredentialStore implements AuthCredentialStore {
|
|
|
70
70
|
getCache(key: string): string | null;
|
|
71
71
|
setCache(key: string, value: string, expiresAtSec: number): void;
|
|
72
72
|
cleanExpiredCache(): void;
|
|
73
|
+
deleteCachePrefix(prefix: string): void;
|
|
73
74
|
/**
|
|
74
75
|
* Store-level hook consumed by `AuthStorage` — routes refresh through the
|
|
75
76
|
* broker so the actual refresh token never leaves the broker host. Returns
|
|
@@ -420,6 +420,7 @@ export declare const credentialIfAbsentUploadResponseSchema: z.ZodObject<{
|
|
|
420
420
|
"skipped-existing-fallback": "skipped-existing-fallback";
|
|
421
421
|
"skipped-existing-runtime": "skipped-existing-runtime";
|
|
422
422
|
"skipped-invalid": "skipped-invalid";
|
|
423
|
+
"updated-existing": "updated-existing";
|
|
423
424
|
}>;
|
|
424
425
|
entries: z.ZodArray<z.ZodObject<{
|
|
425
426
|
id: z.ZodNumber;
|
|
@@ -102,7 +102,7 @@ export interface AuthCredentialSnapshotEntry {
|
|
|
102
102
|
credential: SnapshotCredential;
|
|
103
103
|
identityKey: string | null;
|
|
104
104
|
}
|
|
105
|
-
export type AuthCredentialIfAbsentReason = "inserted" | "skipped-existing" | "skipped-existing-runtime" | "skipped-existing-config" | "skipped-existing-env" | "skipped-existing-fallback" | "skipped-invalid";
|
|
105
|
+
export type AuthCredentialIfAbsentReason = "inserted" | "updated-existing" | "skipped-existing" | "skipped-existing-runtime" | "skipped-existing-config" | "skipped-existing-env" | "skipped-existing-fallback" | "skipped-invalid";
|
|
106
106
|
export interface AuthCredentialIfAbsentResult {
|
|
107
107
|
inserted: boolean;
|
|
108
108
|
reason: AuthCredentialIfAbsentReason;
|
|
@@ -147,6 +147,7 @@ export interface AuthCredentialStore {
|
|
|
147
147
|
includeExpired?: boolean;
|
|
148
148
|
}): string | null;
|
|
149
149
|
setCache(key: string, value: string, expiresAtSec: number): void;
|
|
150
|
+
deleteCachePrefix?(prefix: string): void;
|
|
150
151
|
cleanExpiredCache(): void;
|
|
151
152
|
/**
|
|
152
153
|
* Optional store-supplied OAuth refresh. When present, `AuthStorage` uses
|
|
@@ -646,6 +647,7 @@ export declare class SqliteAuthCredentialStore implements AuthCredentialStore {
|
|
|
646
647
|
includeExpired?: boolean;
|
|
647
648
|
}): string | null;
|
|
648
649
|
setCache(key: string, value: string, expiresAtSec: number): void;
|
|
650
|
+
deleteCachePrefix(prefix: string): void;
|
|
649
651
|
cleanExpiredCache(): void;
|
|
650
652
|
/**
|
|
651
653
|
* Save OAuth credentials for a provider.
|
|
@@ -58,6 +58,11 @@ export interface DeepSeekModelManagerConfig {
|
|
|
58
58
|
baseUrl?: string;
|
|
59
59
|
}
|
|
60
60
|
export declare function deepseekModelManagerOptions(config?: DeepSeekModelManagerConfig): ModelManagerOptions<"openai-completions">;
|
|
61
|
+
export interface DeepInfraModelManagerConfig {
|
|
62
|
+
apiKey?: string;
|
|
63
|
+
baseUrl?: string;
|
|
64
|
+
}
|
|
65
|
+
export declare function deepinfraModelManagerOptions(config?: DeepInfraModelManagerConfig): ModelManagerOptions<"openai-completions">;
|
|
61
66
|
export interface FireworksModelManagerConfig {
|
|
62
67
|
apiKey?: string;
|
|
63
68
|
baseUrl?: string;
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The Cursor client version reported to api2.cursor.sh.
|
|
3
|
+
*
|
|
4
|
+
* Every call against the Cursor backend must send the same
|
|
5
|
+
* x-cursor-client-version: the backend gates features and minimum versions on
|
|
6
|
+
* it, so a drift between the agent Run path and model discovery makes one of
|
|
7
|
+
* them fail while the other keeps working. Keep this as the single source of
|
|
8
|
+
* truth for the header value.
|
|
9
|
+
*/
|
|
10
|
+
export declare const CURSOR_CLIENT_VERSION = "cli-2026.02.13-41ac335";
|
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
import { type JsonValue } from "@bufbuild/protobuf";
|
|
2
2
|
import type { CursorExecHandlerResult, CursorExecHandlers, CursorToolResultHandler, Message, StreamFunction, StreamOptions, ToolResultMessage } from "../types";
|
|
3
|
+
import { CURSOR_CLIENT_VERSION } from "./cursor/client-version";
|
|
3
4
|
export declare const CURSOR_API_URL = "https://api2.cursor.sh";
|
|
4
|
-
export
|
|
5
|
+
export { CURSOR_CLIENT_VERSION };
|
|
5
6
|
/** Drop all cached state + blob bytes for a conversation (F15 bound + session-teardown hook). */
|
|
6
7
|
export declare function disposeCursorConversation(conversationId: string): void;
|
|
7
8
|
export interface CursorOptions extends StreamOptions {
|
|
@@ -0,0 +1,3 @@
|
|
|
1
|
+
import type { FetchImpl } from "../types";
|
|
2
|
+
export declare function isOpenAIUsageExhaustionResponse(bodyText: string, retryAfterMs: number | undefined, retryDelayCapMs: number): boolean;
|
|
3
|
+
export declare function wrapOpenAIFetchForBoundedRateLimits(baseFetch: FetchImpl, maxRetryDelayMs: number | undefined): FetchImpl;
|
package/dist/types/types.d.ts
CHANGED
|
@@ -48,7 +48,7 @@ export interface ThinkingConfig {
|
|
|
48
48
|
/** Provider-specific transport used to encode the selected effort. */
|
|
49
49
|
mode: ThinkingControlMode;
|
|
50
50
|
}
|
|
51
|
-
export type KnownProvider = "alibaba-coding-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
|
|
51
|
+
export type KnownProvider = "alibaba-coding-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
|
|
52
52
|
export type Provider = KnownProvider | string;
|
|
53
53
|
import type { Effort } from "./model-thinking";
|
|
54
54
|
/** Token budgets for each thinking level (token-based providers only) */
|
|
@@ -106,9 +106,9 @@ export type ResolvedServiceTier = Exclude<ServiceTier, "openai-only" | "claude-o
|
|
|
106
106
|
export declare function resolveServiceTier(serviceTier: ServiceTier | null | undefined, provider: Provider | undefined): ResolvedServiceTier | undefined;
|
|
107
107
|
/**
|
|
108
108
|
* True when the (possibly scoped) tier should be sent as OpenAI's
|
|
109
|
-
* `service_tier` request field for the given provider.
|
|
110
|
-
*
|
|
111
|
-
* mismatches
|
|
109
|
+
* `service_tier` request field for the given provider. OpenAI accepts
|
|
110
|
+
* `flex`, `scale`, and `priority`; DeepInfra accepts `priority`.
|
|
111
|
+
* Unsupported tiers (`"auto"`, `"default"`) and scope mismatches return false.
|
|
112
112
|
*/
|
|
113
113
|
export declare function shouldSendServiceTier(serviceTier: ServiceTier | null | undefined, provider: Provider | undefined): boolean;
|
|
114
114
|
/**
|
|
@@ -7,7 +7,7 @@ export type OAuthCredentials = {
|
|
|
7
7
|
email?: string;
|
|
8
8
|
accountId?: string;
|
|
9
9
|
};
|
|
10
|
-
export type OAuthProvider = "alibaba-coding-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
|
|
10
|
+
export type OAuthProvider = "alibaba-coding-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
|
|
11
11
|
export type OAuthProviderId = OAuthProvider | (string & {});
|
|
12
12
|
export type OAuthPrompt = {
|
|
13
13
|
message: string;
|
package/package.json
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@sayknow-cli/ai",
|
|
4
|
-
"version": "0.3.
|
|
4
|
+
"version": "0.3.9",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
|
-
"homepage": "https://
|
|
6
|
+
"homepage": "https://sayknow-cli.com",
|
|
7
7
|
"author": "jaybeyond",
|
|
8
8
|
"contributors": [
|
|
9
9
|
"Mario Zechner"
|
|
@@ -43,7 +43,7 @@
|
|
|
43
43
|
"dependencies": {
|
|
44
44
|
"@anthropic-ai/sdk": "^0.94.0",
|
|
45
45
|
"@bufbuild/protobuf": "^2.12.0",
|
|
46
|
-
"@sayknow-cli/utils": "0.3.
|
|
46
|
+
"@sayknow-cli/utils": "0.3.9",
|
|
47
47
|
"openai": "^6.36.0",
|
|
48
48
|
"partial-json": "^0.1.7",
|
|
49
49
|
"zod": "4.4.3"
|
|
@@ -86,6 +86,7 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
|
|
|
86
86
|
#cache: Map<string, CacheEntry> = new Map();
|
|
87
87
|
#usageCache?: UsageCacheEntry;
|
|
88
88
|
#usageInflight?: Promise<UsageReport[] | null>;
|
|
89
|
+
#usageCacheEpoch = 0;
|
|
89
90
|
#closed = false;
|
|
90
91
|
/**
|
|
91
92
|
* `true` once the SSE consumer received its first frame and hasn't dropped
|
|
@@ -465,6 +466,19 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
|
|
|
465
466
|
}
|
|
466
467
|
}
|
|
467
468
|
|
|
469
|
+
deleteCachePrefix(prefix: string): void {
|
|
470
|
+
for (const key of this.#cache.keys()) {
|
|
471
|
+
if (key.startsWith(prefix)) this.#cache.delete(key);
|
|
472
|
+
}
|
|
473
|
+
if (prefix.startsWith("usage_cache:")) this.#invalidateUsageCache();
|
|
474
|
+
}
|
|
475
|
+
|
|
476
|
+
#invalidateUsageCache(): void {
|
|
477
|
+
this.#usageCache = undefined;
|
|
478
|
+
this.#usageInflight = undefined;
|
|
479
|
+
this.#usageCacheEpoch += 1;
|
|
480
|
+
}
|
|
481
|
+
|
|
468
482
|
/**
|
|
469
483
|
* Store-level hook consumed by `AuthStorage` — routes refresh through the
|
|
470
484
|
* broker so the actual refresh token never leaves the broker host. Returns
|
|
@@ -561,10 +575,11 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
|
|
|
561
575
|
return Promise.resolve(cached.reports);
|
|
562
576
|
}
|
|
563
577
|
if (this.#usageInflight) return this.#usageInflight;
|
|
578
|
+
const epoch = this.#usageCacheEpoch;
|
|
564
579
|
const inflight = this.#client
|
|
565
580
|
.fetchUsage()
|
|
566
581
|
.then(body => {
|
|
567
|
-
this.#usageCache = { reports: body.reports, fetchedAt: Date.now() };
|
|
582
|
+
if (this.#usageCacheEpoch === epoch) this.#usageCache = { reports: body.reports, fetchedAt: Date.now() };
|
|
568
583
|
return body.reports;
|
|
569
584
|
})
|
|
570
585
|
.catch(error => {
|
|
@@ -572,7 +587,7 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
|
|
|
572
587
|
return null;
|
|
573
588
|
})
|
|
574
589
|
.finally(() => {
|
|
575
|
-
this.#usageInflight = undefined;
|
|
590
|
+
if (this.#usageCacheEpoch === epoch) this.#usageInflight = undefined;
|
|
576
591
|
});
|
|
577
592
|
this.#usageInflight = inflight;
|
|
578
593
|
return inflight;
|
|
@@ -187,6 +187,7 @@ export const credentialDisableResponseSchema = z
|
|
|
187
187
|
// ─── Upload ────────────────────────────────────────────────────────────────
|
|
188
188
|
const credentialIfAbsentReasonSchema = z.enum([
|
|
189
189
|
"inserted",
|
|
190
|
+
"updated-existing",
|
|
190
191
|
"skipped-existing",
|
|
191
192
|
"skipped-existing-runtime",
|
|
192
193
|
"skipped-existing-config",
|
package/src/auth-storage.ts
CHANGED
|
@@ -32,6 +32,7 @@ import { kimiUsageProvider } from "./usage/kimi";
|
|
|
32
32
|
import { codexRankingStrategy, openaiCodexUsageProvider } from "./usage/openai-codex";
|
|
33
33
|
import { zaiUsageProvider } from "./usage/zai";
|
|
34
34
|
import { getOAuthApiKey, getOAuthProvider, refreshOAuthToken, resolveOAuthStorageProvider } from "./utils/oauth";
|
|
35
|
+
import { loginDeepInfra } from "./utils/oauth/deepinfra";
|
|
35
36
|
import { loginDeepSeek } from "./utils/oauth/deepseek";
|
|
36
37
|
import { loginOpenAICodexDevice } from "./utils/oauth/openai-codex";
|
|
37
38
|
import type { OAuthController, OAuthCredentials, OAuthProvider, OAuthProviderId } from "./utils/oauth/types";
|
|
@@ -153,6 +154,7 @@ export interface AuthCredentialSnapshotEntry {
|
|
|
153
154
|
|
|
154
155
|
export type AuthCredentialIfAbsentReason =
|
|
155
156
|
| "inserted"
|
|
157
|
+
| "updated-existing"
|
|
156
158
|
| "skipped-existing"
|
|
157
159
|
| "skipped-existing-runtime"
|
|
158
160
|
| "skipped-existing-config"
|
|
@@ -209,6 +211,7 @@ export interface AuthCredentialStore {
|
|
|
209
211
|
deleteAuthCredentialsForProvider(provider: string, disabledCause: string): void;
|
|
210
212
|
getCache(key: string, options?: { includeExpired?: boolean }): string | null;
|
|
211
213
|
setCache(key: string, value: string, expiresAtSec: number): void;
|
|
214
|
+
deleteCachePrefix?(prefix: string): void;
|
|
212
215
|
cleanExpiredCache(): void;
|
|
213
216
|
/**
|
|
214
217
|
* Optional store-supplied OAuth refresh. When present, `AuthStorage` uses
|
|
@@ -471,6 +474,7 @@ interface UsageCache {
|
|
|
471
474
|
get<T>(key: string): UsageCacheEntry<T> | undefined;
|
|
472
475
|
getStale<T>(key: string): UsageCacheEntry<T> | undefined;
|
|
473
476
|
set<T>(key: string, entry: UsageCacheEntry<T>): void;
|
|
477
|
+
deletePrefix?(prefix: string): void;
|
|
474
478
|
cleanup?(): void;
|
|
475
479
|
}
|
|
476
480
|
|
|
@@ -661,6 +665,10 @@ class AuthStorageUsageCache implements UsageCache {
|
|
|
661
665
|
this.store.setCache(`${USAGE_CACHE_PREFIX}${key}`, payload, Math.floor(durableExpiresAt / 1000));
|
|
662
666
|
}
|
|
663
667
|
|
|
668
|
+
deletePrefix(prefix: string): void {
|
|
669
|
+
this.store.deleteCachePrefix?.(`${USAGE_CACHE_PREFIX}${prefix}`);
|
|
670
|
+
}
|
|
671
|
+
|
|
664
672
|
cleanup(): void {
|
|
665
673
|
this.store.cleanExpiredCache();
|
|
666
674
|
}
|
|
@@ -1296,8 +1304,6 @@ export class AuthStorage {
|
|
|
1296
1304
|
return this.#snapshotSkipResult(storageProvider, "skipped-existing-runtime");
|
|
1297
1305
|
if (this.#configOverrides.has(storageProvider))
|
|
1298
1306
|
return this.#snapshotSkipResult(storageProvider, "skipped-existing-config");
|
|
1299
|
-
if (this.#getCredentialsForProvider(storageProvider).length > 0)
|
|
1300
|
-
return this.#snapshotSkipResult(storageProvider, "skipped-existing");
|
|
1301
1307
|
if (getEnvApiKey(storageProvider)) return this.#snapshotSkipResult(storageProvider, "skipped-existing-env");
|
|
1302
1308
|
if (this.#fallbackResolver?.(storageProvider))
|
|
1303
1309
|
return this.#snapshotSkipResult(storageProvider, "skipped-existing-fallback");
|
|
@@ -1310,6 +1316,7 @@ export class AuthStorage {
|
|
|
1310
1316
|
result.entries.map(entry => ({ id: entry.id, credential: entry.credential })),
|
|
1311
1317
|
);
|
|
1312
1318
|
this.#resetProviderAssignments(storageProvider);
|
|
1319
|
+
if (result.inserted) this.#invalidateUsageCacheForProvider(storageProvider);
|
|
1313
1320
|
return {
|
|
1314
1321
|
inserted: result.inserted,
|
|
1315
1322
|
reason: result.reason,
|
|
@@ -1327,6 +1334,13 @@ export class AuthStorage {
|
|
|
1327
1334
|
stored.map(record => ({ id: record.id, credential: record.credential })),
|
|
1328
1335
|
);
|
|
1329
1336
|
this.#resetProviderAssignments(provider);
|
|
1337
|
+
this.#invalidateUsageCacheForProvider(provider);
|
|
1338
|
+
}
|
|
1339
|
+
|
|
1340
|
+
#invalidateUsageCacheForProvider(provider: string): void {
|
|
1341
|
+
this.#usageRequestInFlight.clear();
|
|
1342
|
+
this.#usageReportsInFlight.clear();
|
|
1343
|
+
this.#usageCache.deletePrefix?.(`report:${provider}:`);
|
|
1330
1344
|
}
|
|
1331
1345
|
|
|
1332
1346
|
/**
|
|
@@ -1593,6 +1607,11 @@ export class AuthStorage {
|
|
|
1593
1607
|
await saveApiKeyCredential(apiKey);
|
|
1594
1608
|
return;
|
|
1595
1609
|
}
|
|
1610
|
+
case "deepinfra": {
|
|
1611
|
+
const apiKey = await loginDeepInfra(ctrl);
|
|
1612
|
+
await saveApiKeyCredential(apiKey);
|
|
1613
|
+
return;
|
|
1614
|
+
}
|
|
1596
1615
|
case "xai": {
|
|
1597
1616
|
const { loginXai } = await import("./utils/oauth/xai");
|
|
1598
1617
|
credentials = await loginXai({
|
|
@@ -3699,6 +3718,7 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
|
|
|
3699
3718
|
#getCacheStmt: Statement;
|
|
3700
3719
|
#getCacheIncludingExpiredStmt: Statement;
|
|
3701
3720
|
#upsertCacheStmt: Statement;
|
|
3721
|
+
#deleteCachePrefixStmt: Statement;
|
|
3702
3722
|
#deleteExpiredCacheStmt: Statement;
|
|
3703
3723
|
#closed = false;
|
|
3704
3724
|
|
|
@@ -3738,6 +3758,7 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
|
|
|
3738
3758
|
this.#upsertCacheStmt = this.#db.prepare(
|
|
3739
3759
|
"INSERT INTO cache (key, value, expires_at) VALUES (?, ?, ?) ON CONFLICT(key) DO UPDATE SET value = excluded.value, expires_at = excluded.expires_at",
|
|
3740
3760
|
);
|
|
3761
|
+
this.#deleteCachePrefixStmt = this.#db.prepare("DELETE FROM cache WHERE substr(key, 1, ?) = ?");
|
|
3741
3762
|
this.#deleteExpiredCacheStmt = this.#db.prepare(`DELETE FROM cache WHERE expires_at <= ${SQLITE_NOW_EPOCH}`);
|
|
3742
3763
|
}
|
|
3743
3764
|
|
|
@@ -4096,16 +4117,58 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
|
|
|
4096
4117
|
}
|
|
4097
4118
|
|
|
4098
4119
|
const writeIfAbsent = this.#db.transaction(
|
|
4099
|
-
(
|
|
4120
|
+
(
|
|
4121
|
+
providerName: string,
|
|
4122
|
+
item: AuthCredential,
|
|
4123
|
+
record: SerializedCredentialRecord,
|
|
4124
|
+
): AuthCredentialIfAbsentResult => {
|
|
4100
4125
|
const existingRows = this.#listActiveByProviderStmt.all(providerName) as AuthRow[];
|
|
4101
|
-
const existing:
|
|
4126
|
+
const existing: Array<{
|
|
4127
|
+
id: number;
|
|
4128
|
+
credential: AuthCredential;
|
|
4129
|
+
identityKey: string | null;
|
|
4130
|
+
}> = [];
|
|
4102
4131
|
for (const row of existingRows) {
|
|
4103
4132
|
const activeCredential = deserializeCredential(row);
|
|
4104
4133
|
if (!activeCredential) continue;
|
|
4105
|
-
existing.push(
|
|
4134
|
+
existing.push({
|
|
4135
|
+
id: row.id,
|
|
4136
|
+
credential: activeCredential,
|
|
4137
|
+
identityKey: resolveRowCredentialIdentityKey(providerName, row),
|
|
4138
|
+
});
|
|
4106
4139
|
}
|
|
4107
4140
|
if (existing.length > 0) {
|
|
4108
|
-
|
|
4141
|
+
let targetId: number | null = null;
|
|
4142
|
+
for (const row of existing) {
|
|
4143
|
+
if (!matchesReplacementCredential(providerName, row.credential, row.identityKey, item)) continue;
|
|
4144
|
+
if (targetId === null) {
|
|
4145
|
+
targetId = row.id;
|
|
4146
|
+
this.#updateStmt.run(record.credentialType, record.data, record.identityKey, row.id);
|
|
4147
|
+
} else {
|
|
4148
|
+
this.#deleteStmt.run("replaced by newer credential", row.id);
|
|
4149
|
+
}
|
|
4150
|
+
}
|
|
4151
|
+
|
|
4152
|
+
if (targetId !== null) {
|
|
4153
|
+
return {
|
|
4154
|
+
inserted: true,
|
|
4155
|
+
reason: "updated-existing",
|
|
4156
|
+
provider: providerName,
|
|
4157
|
+
entries: this.listAuthCredentials(providerName),
|
|
4158
|
+
};
|
|
4159
|
+
}
|
|
4160
|
+
|
|
4161
|
+
return {
|
|
4162
|
+
inserted: false,
|
|
4163
|
+
reason: "skipped-existing",
|
|
4164
|
+
provider: providerName,
|
|
4165
|
+
entries: existing.map(row => ({
|
|
4166
|
+
id: row.id,
|
|
4167
|
+
provider: providerName,
|
|
4168
|
+
credential: row.credential,
|
|
4169
|
+
disabledCause: null,
|
|
4170
|
+
})),
|
|
4171
|
+
};
|
|
4109
4172
|
}
|
|
4110
4173
|
|
|
4111
4174
|
this.#insertStmt.get(providerName, record.credentialType, record.data, record.identityKey);
|
|
@@ -4118,7 +4181,9 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
|
|
|
4118
4181
|
},
|
|
4119
4182
|
);
|
|
4120
4183
|
|
|
4121
|
-
|
|
4184
|
+
const result = writeIfAbsent.immediate(provider, credential, serialized);
|
|
4185
|
+
if (result.inserted) this.#purgeSupersededDisabledRows(provider, result.entries);
|
|
4186
|
+
return result;
|
|
4122
4187
|
}
|
|
4123
4188
|
|
|
4124
4189
|
/**
|
|
@@ -4215,6 +4280,13 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
|
|
|
4215
4280
|
}
|
|
4216
4281
|
}
|
|
4217
4282
|
|
|
4283
|
+
deleteCachePrefix(prefix: string): void {
|
|
4284
|
+
if (prefix.length === 0) return;
|
|
4285
|
+
try {
|
|
4286
|
+
this.#deleteCachePrefixStmt.run(prefix.length, prefix);
|
|
4287
|
+
} catch {}
|
|
4288
|
+
}
|
|
4289
|
+
|
|
4218
4290
|
cleanExpiredCache(): void {
|
|
4219
4291
|
try {
|
|
4220
4292
|
this.#deleteExpiredCacheStmt.run();
|
|
@@ -4305,6 +4377,7 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
|
|
|
4305
4377
|
this.#getCacheStmt.finalize();
|
|
4306
4378
|
this.#getCacheIncludingExpiredStmt.finalize();
|
|
4307
4379
|
this.#upsertCacheStmt.finalize();
|
|
4380
|
+
this.#deleteCachePrefixStmt.finalize();
|
|
4308
4381
|
this.#deleteExpiredCacheStmt.finalize();
|
|
4309
4382
|
this.#db.close();
|
|
4310
4383
|
}
|
|
@@ -13,6 +13,7 @@ import {
|
|
|
13
13
|
anthropicModelManagerOptions,
|
|
14
14
|
cerebrasModelManagerOptions,
|
|
15
15
|
cloudflareAiGatewayModelManagerOptions,
|
|
16
|
+
deepinfraModelManagerOptions,
|
|
16
17
|
deepseekModelManagerOptions,
|
|
17
18
|
firepassModelManagerOptions,
|
|
18
19
|
fireworksModelManagerOptions,
|
|
@@ -168,6 +169,12 @@ export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [
|
|
|
168
169
|
config => deepseekModelManagerOptions(config),
|
|
169
170
|
catalog("DeepSeek", ["DEEPSEEK_API_KEY"]),
|
|
170
171
|
),
|
|
172
|
+
catalogDescriptor(
|
|
173
|
+
"deepinfra",
|
|
174
|
+
"deepseek-ai/DeepSeek-V3.2",
|
|
175
|
+
config => deepinfraModelManagerOptions(config),
|
|
176
|
+
catalog("DeepInfra", ["DEEPINFRA_API_KEY"]),
|
|
177
|
+
),
|
|
171
178
|
descriptor("mistral", "devstral-medium-latest", config => mistralModelManagerOptions(config)),
|
|
172
179
|
catalogDescriptor(
|
|
173
180
|
"nvidia",
|
|
@@ -665,6 +665,21 @@ export function deepseekModelManagerOptions(
|
|
|
665
665
|
): ModelManagerOptions<"openai-completions"> {
|
|
666
666
|
return createSimpleOpenAICompletionsOptions("deepseek", "https://api.deepseek.com", config);
|
|
667
667
|
}
|
|
668
|
+
export interface DeepInfraModelManagerConfig {
|
|
669
|
+
apiKey?: string;
|
|
670
|
+
baseUrl?: string;
|
|
671
|
+
}
|
|
672
|
+
|
|
673
|
+
export function deepinfraModelManagerOptions(
|
|
674
|
+
config?: DeepInfraModelManagerConfig,
|
|
675
|
+
): ModelManagerOptions<"openai-completions"> {
|
|
676
|
+
return createSimpleOpenAICompletionsOptions(
|
|
677
|
+
"deepinfra" as Parameters<typeof getBundledModels>[0],
|
|
678
|
+
"https://api.deepinfra.com/v1/openai",
|
|
679
|
+
config,
|
|
680
|
+
);
|
|
681
|
+
}
|
|
682
|
+
|
|
668
683
|
// ---------------------------------------------------------------------------
|
|
669
684
|
// 7.5 Fireworks
|
|
670
685
|
// ---------------------------------------------------------------------------
|
|
@@ -2254,6 +2269,8 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
|
|
|
2254
2269
|
}),
|
|
2255
2270
|
// --- xAI ---
|
|
2256
2271
|
openAiCompletionsDescriptor("xai", "xai", "https://api.x.ai/v1"),
|
|
2272
|
+
// --- DeepInfra ---
|
|
2273
|
+
openAiCompletionsDescriptor("deepinfra", "deepinfra", "https://api.deepinfra.com/v1/openai"),
|
|
2257
2274
|
// --- DeepSeek ---
|
|
2258
2275
|
openAiCompletionsDescriptor("deepseek", "deepseek", "https://api.deepseek.com", {
|
|
2259
2276
|
// Only ship the v4 family as built-ins; older deepseek-chat / deepseek-reasoner
|
|
@@ -35,6 +35,7 @@ import {
|
|
|
35
35
|
markToolChoiceIncapability,
|
|
36
36
|
resolveToolChoice,
|
|
37
37
|
} from "../utils/tool-choice-capability";
|
|
38
|
+
import { wrapOpenAIFetchForBoundedRateLimits } from "./openai-bounded-rate-limits";
|
|
38
39
|
import { normalizeOpenAIResponsesPromptCacheKey, supportsDeveloperRole } from "./openai-responses";
|
|
39
40
|
import {
|
|
40
41
|
appendResponsesToolResultMessages,
|
|
@@ -272,7 +273,7 @@ function createClient(model: Model<"azure-openai-responses">, apiKey: string, op
|
|
|
272
273
|
|
|
273
274
|
const { baseUrl, apiVersion } = resolveAzureConfig(model, options);
|
|
274
275
|
|
|
275
|
-
const baseFetch = options?.fetch ?? fetch;
|
|
276
|
+
const baseFetch = wrapOpenAIFetchForBoundedRateLimits(options?.fetch ?? fetch, options?.maxRetryDelayMs);
|
|
276
277
|
const onSseEvent = options?.onSseEvent;
|
|
277
278
|
return new AzureOpenAI({
|
|
278
279
|
apiKey,
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The Cursor client version reported to api2.cursor.sh.
|
|
3
|
+
*
|
|
4
|
+
* Every call against the Cursor backend must send the same
|
|
5
|
+
* x-cursor-client-version: the backend gates features and minimum versions on
|
|
6
|
+
* it, so a drift between the agent Run path and model discovery makes one of
|
|
7
|
+
* them fail while the other keeps working. Keep this as the single source of
|
|
8
|
+
* truth for the header value.
|
|
9
|
+
*/
|
|
10
|
+
export const CURSOR_CLIENT_VERSION = "cli-2026.02.13-41ac335";
|
package/src/providers/cursor.ts
CHANGED
|
@@ -31,6 +31,7 @@ import { parseStreamingJson } from "../utils/json-parse";
|
|
|
31
31
|
import { formatErrorMessageWithRetryAfter } from "../utils/retry-after";
|
|
32
32
|
import { flattenToolRootCombinators, toolWireSchema } from "../utils/schema";
|
|
33
33
|
import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
|
|
34
|
+
import { CURSOR_CLIENT_VERSION } from "./cursor/client-version";
|
|
34
35
|
import type { McpToolDefinition } from "./cursor/gen/agent_pb";
|
|
35
36
|
import {
|
|
36
37
|
AgentClientMessageSchema,
|
|
@@ -131,7 +132,7 @@ import {
|
|
|
131
132
|
} from "./cursor/gen/agent_pb";
|
|
132
133
|
|
|
133
134
|
export const CURSOR_API_URL = "https://api2.cursor.sh";
|
|
134
|
-
export
|
|
135
|
+
export { CURSOR_CLIENT_VERSION };
|
|
135
136
|
|
|
136
137
|
const conversationStateCache = new Map<string, ConversationStateStructure>();
|
|
137
138
|
const conversationBlobStores = new Map<string, Map<string, Uint8Array>>();
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
* GeminiCLI/VERSION/MODEL (PLATFORM; ARCH; SURFACE)
|
|
5
5
|
*/
|
|
6
6
|
export function getGeminiCliUserAgent(modelId = "gemini-3.1-pro-preview"): string {
|
|
7
|
-
const version = process.env.PI_AI_GEMINI_CLI_VERSION || "0.
|
|
7
|
+
const version = process.env.SKC_AI_GEMINI_CLI_VERSION || process.env.PI_AI_GEMINI_CLI_VERSION || "0.49.0";
|
|
8
8
|
const platform = process.platform === "win32" ? "win32" : process.platform;
|
|
9
9
|
const arch = process.arch === "x64" ? "x64" : process.arch;
|
|
10
10
|
return `GeminiCLI/${version}/${modelId} (${platform}; ${arch}; terminal)`;
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
import type { FetchImpl } from "../types";
|
|
2
|
+
import { getRetryAfterMsFromHeaders } from "../utils/retry-after";
|
|
3
|
+
|
|
4
|
+
const OPENAI_RETRY_DELAY_CAP_MS = 60_000;
|
|
5
|
+
|
|
6
|
+
// Mirror of `wrapAnthropicFetchForBoundedRateLimits`: OpenAI-compatible providers
|
|
7
|
+
// (e.g. opencode-go) return HTTP 429 for *permanent* usage/quota exhaustion — a
|
|
8
|
+
// monthly-cap reset that can be days away. The OpenAI SDK treats 429 as transient
|
|
9
|
+
// and retries up to `maxRetries`, honoring an out-of-range `Retry-After`; the
|
|
10
|
+
// `create()` call then hangs before the error can surface to the agent loop, so
|
|
11
|
+
// no assistant error is produced and the session-level retry/fallback never runs.
|
|
12
|
+
// Detect exhaustion and set `x-should-retry: false` so the SDK gives up at once
|
|
13
|
+
// and the session retry layer applies its own fail-fast (retry-after > maxDelayMs).
|
|
14
|
+
//
|
|
15
|
+
// Shared by every adapter that drives a raw OpenAI SDK client — openai-completions,
|
|
16
|
+
// openai-responses, and azure-openai-responses. Adapters that route through
|
|
17
|
+
// `fetchWithRetry` (codex, bedrock, ollama, gemini-cli) already bound 429 retries
|
|
18
|
+
// themselves and do not need this wrapper.
|
|
19
|
+
export function isOpenAIUsageExhaustionResponse(
|
|
20
|
+
bodyText: string,
|
|
21
|
+
retryAfterMs: number | undefined,
|
|
22
|
+
retryDelayCapMs: number,
|
|
23
|
+
): boolean {
|
|
24
|
+
if (retryAfterMs !== undefined && retryAfterMs > retryDelayCapMs) return true;
|
|
25
|
+
return /monthly usage limit|usage limit reached|usage_limit_reached|out_of_credits|insufficient_quota|quota[ _]?exceeded/i.test(
|
|
26
|
+
bodyText,
|
|
27
|
+
);
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export function wrapOpenAIFetchForBoundedRateLimits(
|
|
31
|
+
baseFetch: FetchImpl,
|
|
32
|
+
maxRetryDelayMs: number | undefined,
|
|
33
|
+
): FetchImpl {
|
|
34
|
+
const retryDelayCapMs = maxRetryDelayMs ?? OPENAI_RETRY_DELAY_CAP_MS;
|
|
35
|
+
return Object.assign(
|
|
36
|
+
async (input: string | URL | Request, init?: RequestInit): Promise<Response> => {
|
|
37
|
+
const response = await baseFetch(input, init);
|
|
38
|
+
if (response.status !== 429 || retryDelayCapMs === 0) return response;
|
|
39
|
+
|
|
40
|
+
const headers = new Headers(response.headers);
|
|
41
|
+
const retryAfterMs = getRetryAfterMsFromHeaders(headers);
|
|
42
|
+
const bodyText = await response
|
|
43
|
+
.clone()
|
|
44
|
+
.text()
|
|
45
|
+
.catch(() => "");
|
|
46
|
+
if (!isOpenAIUsageExhaustionResponse(bodyText, retryAfterMs, retryDelayCapMs)) return response;
|
|
47
|
+
|
|
48
|
+
headers.set("x-should-retry", "false");
|
|
49
|
+
return new Response(bodyText, {
|
|
50
|
+
status: response.status,
|
|
51
|
+
statusText: response.statusText,
|
|
52
|
+
headers,
|
|
53
|
+
});
|
|
54
|
+
},
|
|
55
|
+
baseFetch.preconnect ? { preconnect: baseFetch.preconnect } : {},
|
|
56
|
+
);
|
|
57
|
+
}
|
|
@@ -104,6 +104,10 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
|
|
104
104
|
baseUrl.includes("opencode.ai");
|
|
105
105
|
const isOpenCodeProvider = provider === "opencode-go" || provider === "opencode-zen";
|
|
106
106
|
const isOpenCodeGoReasoning = provider === "opencode-go" && Boolean(model.reasoning);
|
|
107
|
+
const isOpenCodeGoKimiReasoning = provider === "opencode-go" && isKimiModel && Boolean(model.reasoning);
|
|
108
|
+
const isOpenCodeGoKimi25Reasoning = isOpenCodeGoKimiReasoning && model.id === "kimi-k2.5";
|
|
109
|
+
const isOpenCodeGoKimi27CodeReasoning = isOpenCodeGoKimiReasoning && model.id === "kimi-k2.7-code";
|
|
110
|
+
const needsOpenCodeGoKimiEffortMap = isOpenCodeGoKimi25Reasoning || isOpenCodeGoKimi27CodeReasoning;
|
|
107
111
|
|
|
108
112
|
const useMaxTokens =
|
|
109
113
|
provider === "mistral" ||
|
|
@@ -170,22 +174,31 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
|
|
170
174
|
xhigh: "default",
|
|
171
175
|
max: "default",
|
|
172
176
|
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
|
|
173
|
-
:
|
|
177
|
+
: needsOpenCodeGoKimiEffortMap
|
|
174
178
|
? ({
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
179
|
+
// Live Go probes (2026-07-06) showed model-specific effort gaps:
|
|
180
|
+
// kimi-k2.5 rejects "minimal", while kimi-k2.7-code rejects
|
|
181
|
+
// OpenAI-style "xhigh" and "max"; all other Kimi efforts tested
|
|
182
|
+
// successfully and should pass through unchanged.
|
|
183
|
+
...(isOpenCodeGoKimi25Reasoning ? { minimal: "low" } : {}),
|
|
184
|
+
...(isOpenCodeGoKimi27CodeReasoning ? { xhigh: "high", max: "high" } : {}),
|
|
181
185
|
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
|
|
182
|
-
:
|
|
186
|
+
: isDeepseekFamily && model.reasoning
|
|
183
187
|
? ({
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
188
|
+
minimal: "high",
|
|
189
|
+
low: "high",
|
|
190
|
+
medium: "high",
|
|
191
|
+
high: "high",
|
|
192
|
+
xhigh: "max",
|
|
193
|
+
max: "max",
|
|
187
194
|
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
|
|
188
|
-
:
|
|
195
|
+
: isFireworks
|
|
196
|
+
? ({
|
|
197
|
+
// Fireworks' OpenAI-compatible endpoint rejects OpenAI's
|
|
198
|
+
// `minimal` literal but accepts `none` for the lowest setting.
|
|
199
|
+
minimal: "none",
|
|
200
|
+
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
|
|
201
|
+
: {};
|
|
189
202
|
|
|
190
203
|
return {
|
|
191
204
|
supportsStore: !isNonStandard,
|
|
@@ -198,7 +211,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
|
|
198
211
|
disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel || isOpenCodeGoReasoning,
|
|
199
212
|
disableReasoningOnToolChoice: isDeepseekFamily && Boolean(model.reasoning) && !isOpenRouter,
|
|
200
213
|
supportsToolChoice: !isDirectDeepseekReasoning,
|
|
201
|
-
supportsForcedToolChoice:
|
|
214
|
+
supportsForcedToolChoice: !isOpenCodeGoKimiReasoning,
|
|
202
215
|
maxTokensField: useMaxTokens ? "max_tokens" : "max_completion_tokens",
|
|
203
216
|
requiresToolResultName: isMistral,
|
|
204
217
|
requiresAssistantAfterToolResult: false,
|
|
@@ -35,7 +35,7 @@ import {
|
|
|
35
35
|
type ToolChoice,
|
|
36
36
|
type ToolResultMessage,
|
|
37
37
|
} from "../types";
|
|
38
|
-
import { normalizeSystemPrompts } from "../utils";
|
|
38
|
+
import { normalizeSystemPrompts, sanitizeJsonStrings } from "../utils";
|
|
39
39
|
import { createAbortSourceTracker } from "../utils/abort";
|
|
40
40
|
import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
41
41
|
import { toFirepassWireModelId, toFireworksWireModelId } from "../utils/fireworks-model-id";
|
|
@@ -166,7 +166,7 @@ function normalizeStreamingContentText(content: unknown): string {
|
|
|
166
166
|
function serializeToolArguments(value: unknown): string {
|
|
167
167
|
if (value && typeof value === "object" && !Array.isArray(value)) {
|
|
168
168
|
try {
|
|
169
|
-
return JSON.stringify(value);
|
|
169
|
+
return JSON.stringify(sanitizeJsonStrings(value));
|
|
170
170
|
} catch {
|
|
171
171
|
return "{}";
|
|
172
172
|
}
|
|
@@ -178,7 +178,7 @@ function serializeToolArguments(value: unknown): string {
|
|
|
178
178
|
try {
|
|
179
179
|
const parsed = JSON.parse(trimmed);
|
|
180
180
|
if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) {
|
|
181
|
-
return JSON.stringify(parsed);
|
|
181
|
+
return JSON.stringify(sanitizeJsonStrings(parsed));
|
|
182
182
|
}
|
|
183
183
|
} catch {}
|
|
184
184
|
return "{}";
|
|
@@ -71,6 +71,7 @@ import {
|
|
|
71
71
|
resolveGitHubCopilotBaseUrl,
|
|
72
72
|
} from "./github-copilot-headers";
|
|
73
73
|
import { compactGrammarDefinition } from "./grammar";
|
|
74
|
+
import { wrapOpenAIFetchForBoundedRateLimits } from "./openai-bounded-rate-limits";
|
|
74
75
|
import {
|
|
75
76
|
applyOpenAIRequestTransformBody,
|
|
76
77
|
applyOpenAIRequestTransformHeaders,
|
|
@@ -274,6 +275,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
|
|
274
275
|
options?.fetch,
|
|
275
276
|
options?.authCredentialType,
|
|
276
277
|
options?.requestMaxRetries,
|
|
278
|
+
options?.maxRetryDelayMs,
|
|
277
279
|
);
|
|
278
280
|
const premiumRequestsTotal = copilotPremiumRequests;
|
|
279
281
|
const providerSessionState = getOpenAIResponsesProviderSessionState(model, options?.providerSessionState);
|
|
@@ -400,6 +402,7 @@ function createClient(
|
|
|
400
402
|
fetchOverride?: FetchImpl,
|
|
401
403
|
authCredentialType?: OpenAIResponsesOptions["authCredentialType"],
|
|
402
404
|
requestMaxRetries?: number,
|
|
405
|
+
maxRetryDelayMs?: number,
|
|
403
406
|
): {
|
|
404
407
|
client: OpenAI;
|
|
405
408
|
copilotPremiumRequests: number | undefined;
|
|
@@ -446,8 +449,9 @@ function createClient(
|
|
|
446
449
|
headers["x-client-request-id"] ??= sessionId;
|
|
447
450
|
}
|
|
448
451
|
const baseFetch = fetchOverride ?? fetch;
|
|
452
|
+
const boundedFetch = wrapOpenAIFetchForBoundedRateLimits(baseFetch, maxRetryDelayMs);
|
|
449
453
|
const transformedFetch = wrapFetchForOpenAIRequestTransform(
|
|
450
|
-
|
|
454
|
+
boundedFetch,
|
|
451
455
|
model.requestTransform,
|
|
452
456
|
`Sayknow-CLI/${packageJson.version}`,
|
|
453
457
|
);
|
package/src/rate-limit-utils.ts
CHANGED
|
@@ -13,7 +13,7 @@ export type RateLimitReason =
|
|
|
13
13
|
const QUOTA_EXHAUSTED_BACKOFF_MS = 30 * 60 * 1000; // 30 min
|
|
14
14
|
const RATE_LIMIT_EXCEEDED_BACKOFF_MS = 30 * 1000; // 30s
|
|
15
15
|
const MODEL_CAPACITY_BASE_MS = 45 * 1000; // 45s base
|
|
16
|
-
const MODEL_CAPACITY_JITTER_MS = 30 * 1000; //
|
|
16
|
+
const MODEL_CAPACITY_JITTER_MS = 30 * 1000; // uniform +0–30s above base → 45–75s total
|
|
17
17
|
const SERVER_ERROR_BACKOFF_MS = 20 * 1000; // 20s
|
|
18
18
|
|
|
19
19
|
/**
|
|
@@ -93,8 +93,10 @@ export function calculateRateLimitBackoffMs(reason: RateLimitReason): number {
|
|
|
93
93
|
}
|
|
94
94
|
|
|
95
95
|
/** Detect usage/quota limit errors in error messages (persistent, requires credential switch). */
|
|
96
|
+
// ZAI reports durable token exhaustion as "[1310][Weekly/Monthly Limit Exhausted...]".
|
|
97
|
+
// Keep this explicit so generic "rate limit exhausted, retry..." throttles remain retryable.
|
|
96
98
|
const USAGE_LIMIT_PATTERN =
|
|
97
|
-
/usage.?limit|usage_limit_reached|usage_not_included|limit_reached|model.?limit|model_limit_reached|message.?limit|message_limit_reached|limit for this model|quota.?exceeded|out_of_credits|request would exceed your account.?s rate limit|resource has been exhausted[^\n]*(?:quota|limit)/i;
|
|
99
|
+
/usage.?limit|usage_limit_reached|usage_not_included|limit_reached|model.?limit|model_limit_reached|message.?limit|message_limit_reached|limit for this model|weekly\/monthly\s+limit\s+exhausted|quota.?exceeded|out_of_credits|request would exceed your account.?s rate limit|resource has been exhausted[^\n]*(?:quota|limit)/i;
|
|
98
100
|
export function isUsageLimitError(errorMessage: string): boolean {
|
|
99
101
|
return USAGE_LIMIT_PATTERN.test(errorMessage);
|
|
100
102
|
}
|
package/src/stream.ts
CHANGED
|
@@ -98,6 +98,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
|
|
|
98
98
|
"opencode-zen": "OPENCODE_API_KEY",
|
|
99
99
|
cursor: "CURSOR_ACCESS_TOKEN",
|
|
100
100
|
deepseek: "DEEPSEEK_API_KEY",
|
|
101
|
+
deepinfra: "DEEPINFRA_API_KEY",
|
|
101
102
|
"openai-codex": "OPENAI_CODEX_OAUTH_TOKEN",
|
|
102
103
|
"azure-openai": "AZURE_OPENAI_API_KEY",
|
|
103
104
|
"azure-openai-responses": "AZURE_OPENAI_API_KEY",
|
package/src/types.ts
CHANGED
|
@@ -116,6 +116,7 @@ export type KnownProvider =
|
|
|
116
116
|
| "gitlab-duo"
|
|
117
117
|
| "cursor"
|
|
118
118
|
| "deepseek"
|
|
119
|
+
| "deepinfra"
|
|
119
120
|
| "xai"
|
|
120
121
|
| "groq"
|
|
121
122
|
| "cerebras"
|
|
@@ -224,17 +225,20 @@ export function resolveServiceTier(
|
|
|
224
225
|
|
|
225
226
|
/**
|
|
226
227
|
* True when the (possibly scoped) tier should be sent as OpenAI's
|
|
227
|
-
* `service_tier` request field for the given provider.
|
|
228
|
-
*
|
|
229
|
-
* mismatches
|
|
228
|
+
* `service_tier` request field for the given provider. OpenAI accepts
|
|
229
|
+
* `flex`, `scale`, and `priority`; DeepInfra accepts `priority`.
|
|
230
|
+
* Unsupported tiers (`"auto"`, `"default"`) and scope mismatches return false.
|
|
230
231
|
*/
|
|
231
232
|
export function shouldSendServiceTier(
|
|
232
233
|
serviceTier: ServiceTier | null | undefined,
|
|
233
234
|
provider: Provider | undefined,
|
|
234
235
|
): boolean {
|
|
235
|
-
if (provider !== "openai" && provider !== "openai-codex") return false;
|
|
236
236
|
const resolved = resolveServiceTier(serviceTier, provider);
|
|
237
|
-
|
|
237
|
+
if (provider === "openai" || provider === "openai-codex") {
|
|
238
|
+
return resolved === "flex" || resolved === "scale" || resolved === "priority";
|
|
239
|
+
}
|
|
240
|
+
if (provider === "deepinfra") return resolved === "priority";
|
|
241
|
+
return false;
|
|
238
242
|
}
|
|
239
243
|
|
|
240
244
|
/**
|
|
@@ -252,7 +256,9 @@ export function getPriorityPremiumRequests(
|
|
|
252
256
|
if (resolveServiceTier(serviceTier, provider) !== "priority") return 0;
|
|
253
257
|
// Only providers that realize `priority` on the wire bill the user.
|
|
254
258
|
// Everywhere else, the field is silently dropped and nothing is charged.
|
|
255
|
-
return provider === "openai" || provider === "openai-codex" || provider === "anthropic"
|
|
259
|
+
return provider === "openai" || provider === "openai-codex" || provider === "anthropic" || provider === "deepinfra"
|
|
260
|
+
? 1
|
|
261
|
+
: 0;
|
|
256
262
|
}
|
|
257
263
|
|
|
258
264
|
export interface ProviderSessionState {
|
|
@@ -2,11 +2,11 @@ import * as http2 from "node:http2";
|
|
|
2
2
|
import { create, fromBinary, toBinary } from "@bufbuild/protobuf";
|
|
3
3
|
import * as z from "zod/v4";
|
|
4
4
|
import { getBundledModels } from "../../models";
|
|
5
|
+
import { CURSOR_CLIENT_VERSION } from "../../providers/cursor/client-version";
|
|
5
6
|
import { GetUsableModelsRequestSchema, GetUsableModelsResponseSchema } from "../../providers/cursor/gen/agent_pb";
|
|
6
7
|
import type { Model } from "../../types";
|
|
7
8
|
|
|
8
9
|
const CURSOR_DEFAULT_BASE_URL = "https://api2.cursor.sh";
|
|
9
|
-
const CURSOR_DEFAULT_CLIENT_VERSION = "cli-2026.02.13-41ac335";
|
|
10
10
|
const CURSOR_GET_USABLE_MODELS_PATH = "/agent.v1.AgentService/GetUsableModels";
|
|
11
11
|
|
|
12
12
|
const DEFAULT_CONTEXT_WINDOW = 200_000;
|
|
@@ -91,7 +91,7 @@ function buildRequestHeaders(options: CursorModelDiscoveryOptions): Record<strin
|
|
|
91
91
|
te: "trailers",
|
|
92
92
|
authorization: `Bearer ${options.apiKey}`,
|
|
93
93
|
"x-ghost-mode": "true",
|
|
94
|
-
"x-cursor-client-version": options.clientVersion ??
|
|
94
|
+
"x-cursor-client-version": options.clientVersion ?? CURSOR_CLIENT_VERSION,
|
|
95
95
|
"x-cursor-client-type": "cli",
|
|
96
96
|
};
|
|
97
97
|
}
|
package/src/utils/oauth/fugu.ts
CHANGED
|
@@ -3,10 +3,10 @@ import { createApiKeyLogin } from "./api-key-login";
|
|
|
3
3
|
|
|
4
4
|
export const loginFugu = createApiKeyLogin({
|
|
5
5
|
providerLabel: "Sakana Fugu",
|
|
6
|
-
authUrl: "https://
|
|
6
|
+
authUrl: "https://console.sakana.ai/api-keys",
|
|
7
7
|
instructions: "Create or copy your Sakana Fugu API key",
|
|
8
8
|
promptMessage: "Paste your Sakana Fugu API key",
|
|
9
|
-
placeholder: "
|
|
9
|
+
placeholder: "fish_...",
|
|
10
10
|
validation: {
|
|
11
11
|
kind: "models-endpoint",
|
|
12
12
|
provider: "Sakana Fugu",
|
package/src/utils/oauth/index.ts
CHANGED
|
@@ -60,6 +60,11 @@ const builtInOAuthProviders: OAuthProviderInfo[] = [
|
|
|
60
60
|
name: "DeepSeek",
|
|
61
61
|
available: true,
|
|
62
62
|
},
|
|
63
|
+
{
|
|
64
|
+
id: "deepinfra",
|
|
65
|
+
name: "DeepInfra",
|
|
66
|
+
available: true,
|
|
67
|
+
},
|
|
63
68
|
{
|
|
64
69
|
id: "xai",
|
|
65
70
|
name: "xAI",
|
|
@@ -359,6 +364,7 @@ export async function refreshOAuthToken(
|
|
|
359
364
|
case "fireworks":
|
|
360
365
|
case "firepass":
|
|
361
366
|
case "fugu":
|
|
367
|
+
case "deepinfra":
|
|
362
368
|
case "nvidia":
|
|
363
369
|
case "nanogpt":
|
|
364
370
|
case "synthetic":
|
package/src/utils/oauth/types.ts
CHANGED
package/src/utils.ts
CHANGED
|
@@ -215,7 +215,11 @@ function sanitizeOpenAIResponsesHistoryItemForReplay(
|
|
|
215
215
|
// providerPayload stores raw output items; replay strips fields that are output-only.
|
|
216
216
|
const { id: _id, ...itemWithoutId } = item;
|
|
217
217
|
const sanitizedItem =
|
|
218
|
-
item.type === "computer_call"
|
|
218
|
+
item.type === "computer_call"
|
|
219
|
+
? sanitizeComputerCallForResponsesInput(itemWithoutId)
|
|
220
|
+
: item.type === "image_generation_call"
|
|
221
|
+
? sanitizeImageGenerationCallForResponsesInput(itemWithoutId)
|
|
222
|
+
: itemWithoutId;
|
|
219
223
|
if (typeof item.call_id === "string") {
|
|
220
224
|
sanitizedItem.call_id = normalizeReplayedResponsesHistoryCallId(item.call_id, normalizedCallIds);
|
|
221
225
|
}
|
|
@@ -231,6 +235,22 @@ function sanitizeComputerCallForResponsesInput(item: Record<string, unknown>): R
|
|
|
231
235
|
return inputSafeItem;
|
|
232
236
|
}
|
|
233
237
|
|
|
238
|
+
function sanitizeImageGenerationCallForResponsesInput(item: Record<string, unknown>): Record<string, unknown> {
|
|
239
|
+
// Image generation output items include request-time knobs that are not part of
|
|
240
|
+
// the Responses input replay schema. Replaying them verbatim makes OpenAI-compatible
|
|
241
|
+
// endpoints reject the next turn, e.g. `Unknown parameter: input[n].action`.
|
|
242
|
+
const {
|
|
243
|
+
action: _action,
|
|
244
|
+
background: _background,
|
|
245
|
+
output_format: _outputFormat,
|
|
246
|
+
quality: _quality,
|
|
247
|
+
revised_prompt: _revisedPrompt,
|
|
248
|
+
size: _size,
|
|
249
|
+
...inputSafeItem
|
|
250
|
+
} = item;
|
|
251
|
+
return inputSafeItem;
|
|
252
|
+
}
|
|
253
|
+
|
|
234
254
|
function normalizeReplayedResponsesHistoryCallId(value: string, normalizedValues: Map<string, string>): string {
|
|
235
255
|
const normalized = normalizedValues.get(value);
|
|
236
256
|
if (normalized) return normalized;
|