@sayknow-cli/ai 0.3.8 → 0.3.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/types/auth-storage.d.ts +16 -0
- package/dist/types/provider-models/openai-compat.d.ts +5 -0
- package/dist/types/providers/cursor/client-version.d.ts +10 -0
- package/dist/types/providers/cursor.d.ts +2 -1
- package/dist/types/types.d.ts +5 -6
- package/dist/types/utils/oauth/types.d.ts +1 -1
- package/package.json +2 -2
- package/src/auth-storage.ts +134 -12
- package/src/models.json +3589 -673
- package/src/provider-models/descriptors.ts +7 -0
- package/src/provider-models/openai-compat.ts +13 -0
- package/src/providers/cursor/client-version.ts +10 -0
- package/src/providers/cursor.ts +2 -1
- package/src/providers/google-gemini-headers.ts +1 -1
- package/src/providers/openai-completions.ts +3 -3
- package/src/rate-limit-utils.ts +4 -2
- package/src/stream.ts +1 -0
- package/src/types.ts +10 -7
- package/src/utils/discovery/cursor.ts +2 -2
- package/src/utils/oauth/fugu.ts +2 -2
- package/src/utils/oauth/index.ts +5 -0
- package/src/utils/oauth/types.ts +1 -0
|
@@ -13,6 +13,7 @@ import {
|
|
|
13
13
|
anthropicModelManagerOptions,
|
|
14
14
|
cerebrasModelManagerOptions,
|
|
15
15
|
cloudflareAiGatewayModelManagerOptions,
|
|
16
|
+
deepinfraModelManagerOptions,
|
|
16
17
|
deepseekModelManagerOptions,
|
|
17
18
|
firepassModelManagerOptions,
|
|
18
19
|
fireworksModelManagerOptions,
|
|
@@ -168,6 +169,12 @@ export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [
|
|
|
168
169
|
config => deepseekModelManagerOptions(config),
|
|
169
170
|
catalog("DeepSeek", ["DEEPSEEK_API_KEY"]),
|
|
170
171
|
),
|
|
172
|
+
catalogDescriptor(
|
|
173
|
+
"deepinfra",
|
|
174
|
+
"deepseek-ai/DeepSeek-V3.2",
|
|
175
|
+
config => deepinfraModelManagerOptions(config),
|
|
176
|
+
catalog("DeepInfra", ["DEEPINFRA_API_KEY"]),
|
|
177
|
+
),
|
|
171
178
|
descriptor("mistral", "devstral-medium-latest", config => mistralModelManagerOptions(config)),
|
|
172
179
|
catalogDescriptor(
|
|
173
180
|
"nvidia",
|
|
@@ -665,6 +665,17 @@ export function deepseekModelManagerOptions(
|
|
|
665
665
|
): ModelManagerOptions<"openai-completions"> {
|
|
666
666
|
return createSimpleOpenAICompletionsOptions("deepseek", "https://api.deepseek.com", config);
|
|
667
667
|
}
|
|
668
|
+
|
|
669
|
+
export interface DeepInfraModelManagerConfig {
|
|
670
|
+
apiKey?: string;
|
|
671
|
+
baseUrl?: string;
|
|
672
|
+
}
|
|
673
|
+
|
|
674
|
+
export function deepinfraModelManagerOptions(
|
|
675
|
+
config?: DeepInfraModelManagerConfig,
|
|
676
|
+
): ModelManagerOptions<"openai-completions"> {
|
|
677
|
+
return createSimpleOpenAICompletionsOptions("deepinfra", "https://api.deepinfra.com/v1/openai", config);
|
|
678
|
+
}
|
|
668
679
|
// ---------------------------------------------------------------------------
|
|
669
680
|
// 7.5 Fireworks
|
|
670
681
|
// ---------------------------------------------------------------------------
|
|
@@ -2282,6 +2293,8 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
|
|
|
2282
2293
|
requiresAssistantContentForToolCalls: true,
|
|
2283
2294
|
},
|
|
2284
2295
|
}),
|
|
2296
|
+
// --- DeepInfra ---
|
|
2297
|
+
openAiCompletionsDescriptor("deepinfra", "deepinfra", "https://api.deepinfra.com/v1/openai"),
|
|
2285
2298
|
];
|
|
2286
2299
|
|
|
2287
2300
|
const MODELS_DEV_PROVIDER_DESCRIPTORS_CODING_PLANS: readonly ModelsDevProviderDescriptor[] = [
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The Cursor client version reported to api2.cursor.sh.
|
|
3
|
+
*
|
|
4
|
+
* Every call against the Cursor backend must send the same
|
|
5
|
+
* x-cursor-client-version: the backend gates features and minimum versions on
|
|
6
|
+
* it, so a drift between the agent Run path and model discovery makes one of
|
|
7
|
+
* them fail while the other keeps working. Keep this as the single source of
|
|
8
|
+
* truth for the header value.
|
|
9
|
+
*/
|
|
10
|
+
export const CURSOR_CLIENT_VERSION = "cli-2026.02.13-41ac335";
|
package/src/providers/cursor.ts
CHANGED
|
@@ -31,6 +31,7 @@ import { parseStreamingJson } from "../utils/json-parse";
|
|
|
31
31
|
import { formatErrorMessageWithRetryAfter } from "../utils/retry-after";
|
|
32
32
|
import { flattenToolRootCombinators, toolWireSchema } from "../utils/schema";
|
|
33
33
|
import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
|
|
34
|
+
import { CURSOR_CLIENT_VERSION } from "./cursor/client-version";
|
|
34
35
|
import type { McpToolDefinition } from "./cursor/gen/agent_pb";
|
|
35
36
|
import {
|
|
36
37
|
AgentClientMessageSchema,
|
|
@@ -131,7 +132,7 @@ import {
|
|
|
131
132
|
} from "./cursor/gen/agent_pb";
|
|
132
133
|
|
|
133
134
|
export const CURSOR_API_URL = "https://api2.cursor.sh";
|
|
134
|
-
export
|
|
135
|
+
export { CURSOR_CLIENT_VERSION };
|
|
135
136
|
|
|
136
137
|
const conversationStateCache = new Map<string, ConversationStateStructure>();
|
|
137
138
|
const conversationBlobStores = new Map<string, Map<string, Uint8Array>>();
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
* GeminiCLI/VERSION/MODEL (PLATFORM; ARCH; SURFACE)
|
|
5
5
|
*/
|
|
6
6
|
export function getGeminiCliUserAgent(modelId = "gemini-3.1-pro-preview"): string {
|
|
7
|
-
const version = process.env.PI_AI_GEMINI_CLI_VERSION || "0.
|
|
7
|
+
const version = process.env.SKC_AI_GEMINI_CLI_VERSION || process.env.PI_AI_GEMINI_CLI_VERSION || "0.49.0";
|
|
8
8
|
const platform = process.platform === "win32" ? "win32" : process.platform;
|
|
9
9
|
const arch = process.arch === "x64" ? "x64" : process.arch;
|
|
10
10
|
return `GeminiCLI/${version}/${modelId} (${platform}; ${arch}; terminal)`;
|
|
@@ -35,7 +35,7 @@ import {
|
|
|
35
35
|
type ToolChoice,
|
|
36
36
|
type ToolResultMessage,
|
|
37
37
|
} from "../types";
|
|
38
|
-
import { normalizeSystemPrompts } from "../utils";
|
|
38
|
+
import { normalizeSystemPrompts, sanitizeJsonStrings } from "../utils";
|
|
39
39
|
import { createAbortSourceTracker } from "../utils/abort";
|
|
40
40
|
import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
41
41
|
import { toFirepassWireModelId, toFireworksWireModelId } from "../utils/fireworks-model-id";
|
|
@@ -166,7 +166,7 @@ function normalizeStreamingContentText(content: unknown): string {
|
|
|
166
166
|
function serializeToolArguments(value: unknown): string {
|
|
167
167
|
if (value && typeof value === "object" && !Array.isArray(value)) {
|
|
168
168
|
try {
|
|
169
|
-
return JSON.stringify(value);
|
|
169
|
+
return JSON.stringify(sanitizeJsonStrings(value));
|
|
170
170
|
} catch {
|
|
171
171
|
return "{}";
|
|
172
172
|
}
|
|
@@ -178,7 +178,7 @@ function serializeToolArguments(value: unknown): string {
|
|
|
178
178
|
try {
|
|
179
179
|
const parsed = JSON.parse(trimmed);
|
|
180
180
|
if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) {
|
|
181
|
-
return JSON.stringify(parsed);
|
|
181
|
+
return JSON.stringify(sanitizeJsonStrings(parsed));
|
|
182
182
|
}
|
|
183
183
|
} catch {}
|
|
184
184
|
return "{}";
|
package/src/rate-limit-utils.ts
CHANGED
|
@@ -13,7 +13,7 @@ export type RateLimitReason =
|
|
|
13
13
|
const QUOTA_EXHAUSTED_BACKOFF_MS = 30 * 60 * 1000; // 30 min
|
|
14
14
|
const RATE_LIMIT_EXCEEDED_BACKOFF_MS = 30 * 1000; // 30s
|
|
15
15
|
const MODEL_CAPACITY_BASE_MS = 45 * 1000; // 45s base
|
|
16
|
-
const MODEL_CAPACITY_JITTER_MS = 30 * 1000; //
|
|
16
|
+
const MODEL_CAPACITY_JITTER_MS = 30 * 1000; // uniform +0–30s above base → 45–75s total
|
|
17
17
|
const SERVER_ERROR_BACKOFF_MS = 20 * 1000; // 20s
|
|
18
18
|
|
|
19
19
|
/**
|
|
@@ -93,8 +93,10 @@ export function calculateRateLimitBackoffMs(reason: RateLimitReason): number {
|
|
|
93
93
|
}
|
|
94
94
|
|
|
95
95
|
/** Detect usage/quota limit errors in error messages (persistent, requires credential switch). */
|
|
96
|
+
// ZAI reports durable token exhaustion as "[1310][Weekly/Monthly Limit Exhausted...]".
|
|
97
|
+
// Keep this explicit so generic "rate limit exhausted, retry..." throttles remain retryable.
|
|
96
98
|
const USAGE_LIMIT_PATTERN =
|
|
97
|
-
/usage.?limit|usage_limit_reached|usage_not_included|limit_reached|model.?limit|model_limit_reached|message.?limit|message_limit_reached|limit for this model|quota.?exceeded|out_of_credits|request would exceed your account.?s rate limit|resource has been exhausted[^\n]*(?:quota|limit)/i;
|
|
99
|
+
/usage.?limit|usage_limit_reached|usage_not_included|limit_reached|model.?limit|model_limit_reached|message.?limit|message_limit_reached|limit for this model|weekly\/monthly\s+limit\s+exhausted|quota.?exceeded|out_of_credits|request would exceed your account.?s rate limit|resource has been exhausted[^\n]*(?:quota|limit)/i;
|
|
98
100
|
export function isUsageLimitError(errorMessage: string): boolean {
|
|
99
101
|
return USAGE_LIMIT_PATTERN.test(errorMessage);
|
|
100
102
|
}
|
package/src/stream.ts
CHANGED
|
@@ -98,6 +98,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
|
|
|
98
98
|
"opencode-zen": "OPENCODE_API_KEY",
|
|
99
99
|
cursor: "CURSOR_ACCESS_TOKEN",
|
|
100
100
|
deepseek: "DEEPSEEK_API_KEY",
|
|
101
|
+
deepinfra: "DEEPINFRA_API_KEY",
|
|
101
102
|
"openai-codex": "OPENAI_CODEX_OAUTH_TOKEN",
|
|
102
103
|
"azure-openai": "AZURE_OPENAI_API_KEY",
|
|
103
104
|
"azure-openai-responses": "AZURE_OPENAI_API_KEY",
|
package/src/types.ts
CHANGED
|
@@ -116,6 +116,7 @@ export type KnownProvider =
|
|
|
116
116
|
| "gitlab-duo"
|
|
117
117
|
| "cursor"
|
|
118
118
|
| "deepseek"
|
|
119
|
+
| "deepinfra"
|
|
119
120
|
| "xai"
|
|
120
121
|
| "groq"
|
|
121
122
|
| "cerebras"
|
|
@@ -186,7 +187,7 @@ export type CacheRetention = "none" | "short" | "long";
|
|
|
186
187
|
*
|
|
187
188
|
* The unscoped values (`"auto"`, `"default"`, `"flex"`, `"scale"`,
|
|
188
189
|
* `"priority"`) are passed through to providers that understand them
|
|
189
|
-
* (OpenAI's `service_tier` field directly; Anthropic translates
|
|
190
|
+
* (OpenAI and DeepInfra's `service_tier` field directly; Anthropic translates
|
|
190
191
|
* `"priority"` into `speed: "fast"` on supported Opus models).
|
|
191
192
|
*
|
|
192
193
|
* The scoped values target a specific provider family and behave as the
|
|
@@ -223,17 +224,17 @@ export function resolveServiceTier(
|
|
|
223
224
|
}
|
|
224
225
|
|
|
225
226
|
/**
|
|
226
|
-
* True when the (possibly scoped) tier should be sent as OpenAI
|
|
227
|
-
* `service_tier` request field for
|
|
228
|
-
*
|
|
229
|
-
* mismatches all return false.
|
|
227
|
+
* True when the (possibly scoped) tier should be sent as an OpenAI-compatible
|
|
228
|
+
* `service_tier` request field for providers that support it. Unsupported tiers
|
|
229
|
+
* (`"auto"`, `"default"`) and scope mismatches all return false.
|
|
230
230
|
*/
|
|
231
231
|
export function shouldSendServiceTier(
|
|
232
232
|
serviceTier: ServiceTier | null | undefined,
|
|
233
233
|
provider: Provider | undefined,
|
|
234
234
|
): boolean {
|
|
235
|
-
if (provider !== "openai" && provider !== "openai-codex") return false;
|
|
236
235
|
const resolved = resolveServiceTier(serviceTier, provider);
|
|
236
|
+
if (provider === "deepinfra") return resolved === "priority";
|
|
237
|
+
if (provider !== "openai" && provider !== "openai-codex") return false;
|
|
237
238
|
return resolved === "flex" || resolved === "scale" || resolved === "priority";
|
|
238
239
|
}
|
|
239
240
|
|
|
@@ -252,7 +253,9 @@ export function getPriorityPremiumRequests(
|
|
|
252
253
|
if (resolveServiceTier(serviceTier, provider) !== "priority") return 0;
|
|
253
254
|
// Only providers that realize `priority` on the wire bill the user.
|
|
254
255
|
// Everywhere else, the field is silently dropped and nothing is charged.
|
|
255
|
-
return provider === "openai" || provider === "openai-codex" || provider === "anthropic"
|
|
256
|
+
return provider === "openai" || provider === "openai-codex" || provider === "anthropic" || provider === "deepinfra"
|
|
257
|
+
? 1
|
|
258
|
+
: 0;
|
|
256
259
|
}
|
|
257
260
|
|
|
258
261
|
export interface ProviderSessionState {
|
|
@@ -2,11 +2,11 @@ import * as http2 from "node:http2";
|
|
|
2
2
|
import { create, fromBinary, toBinary } from "@bufbuild/protobuf";
|
|
3
3
|
import * as z from "zod/v4";
|
|
4
4
|
import { getBundledModels } from "../../models";
|
|
5
|
+
import { CURSOR_CLIENT_VERSION } from "../../providers/cursor/client-version";
|
|
5
6
|
import { GetUsableModelsRequestSchema, GetUsableModelsResponseSchema } from "../../providers/cursor/gen/agent_pb";
|
|
6
7
|
import type { Model } from "../../types";
|
|
7
8
|
|
|
8
9
|
const CURSOR_DEFAULT_BASE_URL = "https://api2.cursor.sh";
|
|
9
|
-
const CURSOR_DEFAULT_CLIENT_VERSION = "cli-2026.02.13-41ac335";
|
|
10
10
|
const CURSOR_GET_USABLE_MODELS_PATH = "/agent.v1.AgentService/GetUsableModels";
|
|
11
11
|
|
|
12
12
|
const DEFAULT_CONTEXT_WINDOW = 200_000;
|
|
@@ -91,7 +91,7 @@ function buildRequestHeaders(options: CursorModelDiscoveryOptions): Record<strin
|
|
|
91
91
|
te: "trailers",
|
|
92
92
|
authorization: `Bearer ${options.apiKey}`,
|
|
93
93
|
"x-ghost-mode": "true",
|
|
94
|
-
"x-cursor-client-version": options.clientVersion ??
|
|
94
|
+
"x-cursor-client-version": options.clientVersion ?? CURSOR_CLIENT_VERSION,
|
|
95
95
|
"x-cursor-client-type": "cli",
|
|
96
96
|
};
|
|
97
97
|
}
|
package/src/utils/oauth/fugu.ts
CHANGED
|
@@ -3,10 +3,10 @@ import { createApiKeyLogin } from "./api-key-login";
|
|
|
3
3
|
|
|
4
4
|
export const loginFugu = createApiKeyLogin({
|
|
5
5
|
providerLabel: "Sakana Fugu",
|
|
6
|
-
authUrl: "https://
|
|
6
|
+
authUrl: "https://console.sakana.ai/api-keys",
|
|
7
7
|
instructions: "Create or copy your Sakana Fugu API key",
|
|
8
8
|
promptMessage: "Paste your Sakana Fugu API key",
|
|
9
|
-
placeholder: "
|
|
9
|
+
placeholder: "fish_...",
|
|
10
10
|
validation: {
|
|
11
11
|
kind: "models-endpoint",
|
|
12
12
|
provider: "Sakana Fugu",
|
package/src/utils/oauth/index.ts
CHANGED