@sayknow-cli/ai 0.3.8 → 0.3.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -58,6 +58,11 @@ export interface DeepSeekModelManagerConfig {
58
58
  baseUrl?: string;
59
59
  }
60
60
  export declare function deepseekModelManagerOptions(config?: DeepSeekModelManagerConfig): ModelManagerOptions<"openai-completions">;
61
+ export interface DeepInfraModelManagerConfig {
62
+ apiKey?: string;
63
+ baseUrl?: string;
64
+ }
65
+ export declare function deepinfraModelManagerOptions(config?: DeepInfraModelManagerConfig): ModelManagerOptions<"openai-completions">;
61
66
  export interface FireworksModelManagerConfig {
62
67
  apiKey?: string;
63
68
  baseUrl?: string;
@@ -0,0 +1,10 @@
1
+ /**
2
+ * The Cursor client version reported to api2.cursor.sh.
3
+ *
4
+ * Every call against the Cursor backend must send the same
5
+ * x-cursor-client-version: the backend gates features and minimum versions on
6
+ * it, so a drift between the agent Run path and model discovery makes one of
7
+ * them fail while the other keeps working. Keep this as the single source of
8
+ * truth for the header value.
9
+ */
10
+ export declare const CURSOR_CLIENT_VERSION = "cli-2026.02.13-41ac335";
@@ -1,7 +1,8 @@
1
1
  import { type JsonValue } from "@bufbuild/protobuf";
2
2
  import type { CursorExecHandlerResult, CursorExecHandlers, CursorToolResultHandler, Message, StreamFunction, StreamOptions, ToolResultMessage } from "../types";
3
+ import { CURSOR_CLIENT_VERSION } from "./cursor/client-version";
3
4
  export declare const CURSOR_API_URL = "https://api2.cursor.sh";
4
- export declare const CURSOR_CLIENT_VERSION = "cli-2026.01.09-231024f";
5
+ export { CURSOR_CLIENT_VERSION };
5
6
  /** Drop all cached state + blob bytes for a conversation (F15 bound + session-teardown hook). */
6
7
  export declare function disposeCursorConversation(conversationId: string): void;
7
8
  export interface CursorOptions extends StreamOptions {
@@ -48,7 +48,7 @@ export interface ThinkingConfig {
48
48
  /** Provider-specific transport used to encode the selected effort. */
49
49
  mode: ThinkingControlMode;
50
50
  }
51
- export type KnownProvider = "alibaba-coding-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
51
+ export type KnownProvider = "alibaba-coding-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
52
52
  export type Provider = KnownProvider | string;
53
53
  import type { Effort } from "./model-thinking";
54
54
  /** Token budgets for each thinking level (token-based providers only) */
@@ -106,9 +106,9 @@ export type ResolvedServiceTier = Exclude<ServiceTier, "openai-only" | "claude-o
106
106
  export declare function resolveServiceTier(serviceTier: ServiceTier | null | undefined, provider: Provider | undefined): ResolvedServiceTier | undefined;
107
107
  /**
108
108
  * True when the (possibly scoped) tier should be sent as OpenAI's
109
- * `service_tier` request field for the given provider. Non-OpenAI
110
- * providers, unsupported tiers (`"auto"`, `"default"`), and scope
111
- * mismatches all return false.
109
+ * `service_tier` request field for the given provider. OpenAI accepts
110
+ * `flex`, `scale`, and `priority`; DeepInfra accepts `priority`.
111
+ * Unsupported tiers (`"auto"`, `"default"`) and scope mismatches return false.
112
112
  */
113
113
  export declare function shouldSendServiceTier(serviceTier: ServiceTier | null | undefined, provider: Provider | undefined): boolean;
114
114
  /**
@@ -7,7 +7,7 @@ export type OAuthCredentials = {
7
7
  email?: string;
8
8
  accountId?: string;
9
9
  };
10
- export type OAuthProvider = "alibaba-coding-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
10
+ export type OAuthProvider = "alibaba-coding-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
11
11
  export type OAuthProviderId = OAuthProvider | (string & {});
12
12
  export type OAuthPrompt = {
13
13
  message: string;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@sayknow-cli/ai",
4
- "version": "0.3.8",
4
+ "version": "0.3.9",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://sayknow-cli.com",
7
7
  "author": "jaybeyond",
@@ -43,7 +43,7 @@
43
43
  "dependencies": {
44
44
  "@anthropic-ai/sdk": "^0.94.0",
45
45
  "@bufbuild/protobuf": "^2.12.0",
46
- "@sayknow-cli/utils": "0.3.8",
46
+ "@sayknow-cli/utils": "0.3.9",
47
47
  "openai": "^6.36.0",
48
48
  "partial-json": "^0.1.7",
49
49
  "zod": "4.4.3"
@@ -13,6 +13,7 @@ import {
13
13
  anthropicModelManagerOptions,
14
14
  cerebrasModelManagerOptions,
15
15
  cloudflareAiGatewayModelManagerOptions,
16
+ deepinfraModelManagerOptions,
16
17
  deepseekModelManagerOptions,
17
18
  firepassModelManagerOptions,
18
19
  fireworksModelManagerOptions,
@@ -168,6 +169,12 @@ export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [
168
169
  config => deepseekModelManagerOptions(config),
169
170
  catalog("DeepSeek", ["DEEPSEEK_API_KEY"]),
170
171
  ),
172
+ catalogDescriptor(
173
+ "deepinfra",
174
+ "deepseek-ai/DeepSeek-V3.2",
175
+ config => deepinfraModelManagerOptions(config),
176
+ catalog("DeepInfra", ["DEEPINFRA_API_KEY"]),
177
+ ),
171
178
  descriptor("mistral", "devstral-medium-latest", config => mistralModelManagerOptions(config)),
172
179
  catalogDescriptor(
173
180
  "nvidia",
@@ -665,6 +665,21 @@ export function deepseekModelManagerOptions(
665
665
  ): ModelManagerOptions<"openai-completions"> {
666
666
  return createSimpleOpenAICompletionsOptions("deepseek", "https://api.deepseek.com", config);
667
667
  }
668
+ export interface DeepInfraModelManagerConfig {
669
+ apiKey?: string;
670
+ baseUrl?: string;
671
+ }
672
+
673
+ export function deepinfraModelManagerOptions(
674
+ config?: DeepInfraModelManagerConfig,
675
+ ): ModelManagerOptions<"openai-completions"> {
676
+ return createSimpleOpenAICompletionsOptions(
677
+ "deepinfra" as Parameters<typeof getBundledModels>[0],
678
+ "https://api.deepinfra.com/v1/openai",
679
+ config,
680
+ );
681
+ }
682
+
668
683
  // ---------------------------------------------------------------------------
669
684
  // 7.5 Fireworks
670
685
  // ---------------------------------------------------------------------------
@@ -2254,6 +2269,8 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
2254
2269
  }),
2255
2270
  // --- xAI ---
2256
2271
  openAiCompletionsDescriptor("xai", "xai", "https://api.x.ai/v1"),
2272
+ // --- DeepInfra ---
2273
+ openAiCompletionsDescriptor("deepinfra", "deepinfra", "https://api.deepinfra.com/v1/openai"),
2257
2274
  // --- DeepSeek ---
2258
2275
  openAiCompletionsDescriptor("deepseek", "deepseek", "https://api.deepseek.com", {
2259
2276
  // Only ship the v4 family as built-ins; older deepseek-chat / deepseek-reasoner
@@ -0,0 +1,10 @@
1
+ /**
2
+ * The Cursor client version reported to api2.cursor.sh.
3
+ *
4
+ * Every call against the Cursor backend must send the same
5
+ * x-cursor-client-version: the backend gates features and minimum versions on
6
+ * it, so a drift between the agent Run path and model discovery makes one of
7
+ * them fail while the other keeps working. Keep this as the single source of
8
+ * truth for the header value.
9
+ */
10
+ export const CURSOR_CLIENT_VERSION = "cli-2026.02.13-41ac335";
@@ -31,6 +31,7 @@ import { parseStreamingJson } from "../utils/json-parse";
31
31
  import { formatErrorMessageWithRetryAfter } from "../utils/retry-after";
32
32
  import { flattenToolRootCombinators, toolWireSchema } from "../utils/schema";
33
33
  import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
34
+ import { CURSOR_CLIENT_VERSION } from "./cursor/client-version";
34
35
  import type { McpToolDefinition } from "./cursor/gen/agent_pb";
35
36
  import {
36
37
  AgentClientMessageSchema,
@@ -131,7 +132,7 @@ import {
131
132
  } from "./cursor/gen/agent_pb";
132
133
 
133
134
  export const CURSOR_API_URL = "https://api2.cursor.sh";
134
- export const CURSOR_CLIENT_VERSION = "cli-2026.01.09-231024f";
135
+ export { CURSOR_CLIENT_VERSION };
135
136
 
136
137
  const conversationStateCache = new Map<string, ConversationStateStructure>();
137
138
  const conversationBlobStores = new Map<string, Map<string, Uint8Array>>();
@@ -4,7 +4,7 @@
4
4
  * GeminiCLI/VERSION/MODEL (PLATFORM; ARCH; SURFACE)
5
5
  */
6
6
  export function getGeminiCliUserAgent(modelId = "gemini-3.1-pro-preview"): string {
7
- const version = process.env.PI_AI_GEMINI_CLI_VERSION || "0.46.0";
7
+ const version = process.env.SKC_AI_GEMINI_CLI_VERSION || process.env.PI_AI_GEMINI_CLI_VERSION || "0.49.0";
8
8
  const platform = process.platform === "win32" ? "win32" : process.platform;
9
9
  const arch = process.arch === "x64" ? "x64" : process.arch;
10
10
  return `GeminiCLI/${version}/${modelId} (${platform}; ${arch}; terminal)`;
@@ -35,7 +35,7 @@ import {
35
35
  type ToolChoice,
36
36
  type ToolResultMessage,
37
37
  } from "../types";
38
- import { normalizeSystemPrompts } from "../utils";
38
+ import { normalizeSystemPrompts, sanitizeJsonStrings } from "../utils";
39
39
  import { createAbortSourceTracker } from "../utils/abort";
40
40
  import { AssistantMessageEventStream } from "../utils/event-stream";
41
41
  import { toFirepassWireModelId, toFireworksWireModelId } from "../utils/fireworks-model-id";
@@ -166,7 +166,7 @@ function normalizeStreamingContentText(content: unknown): string {
166
166
  function serializeToolArguments(value: unknown): string {
167
167
  if (value && typeof value === "object" && !Array.isArray(value)) {
168
168
  try {
169
- return JSON.stringify(value);
169
+ return JSON.stringify(sanitizeJsonStrings(value));
170
170
  } catch {
171
171
  return "{}";
172
172
  }
@@ -178,7 +178,7 @@ function serializeToolArguments(value: unknown): string {
178
178
  try {
179
179
  const parsed = JSON.parse(trimmed);
180
180
  if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) {
181
- return JSON.stringify(parsed);
181
+ return JSON.stringify(sanitizeJsonStrings(parsed));
182
182
  }
183
183
  } catch {}
184
184
  return "{}";
@@ -13,7 +13,7 @@ export type RateLimitReason =
13
13
  const QUOTA_EXHAUSTED_BACKOFF_MS = 30 * 60 * 1000; // 30 min
14
14
  const RATE_LIMIT_EXCEEDED_BACKOFF_MS = 30 * 1000; // 30s
15
15
  const MODEL_CAPACITY_BASE_MS = 45 * 1000; // 45s base
16
- const MODEL_CAPACITY_JITTER_MS = 30 * 1000; // ±15s
16
+ const MODEL_CAPACITY_JITTER_MS = 30 * 1000; // uniform +0–30s above base → 45–75s total
17
17
  const SERVER_ERROR_BACKOFF_MS = 20 * 1000; // 20s
18
18
 
19
19
  /**
@@ -93,8 +93,10 @@ export function calculateRateLimitBackoffMs(reason: RateLimitReason): number {
93
93
  }
94
94
 
95
95
  /** Detect usage/quota limit errors in error messages (persistent, requires credential switch). */
96
+ // ZAI reports durable token exhaustion as "[1310][Weekly/Monthly Limit Exhausted...]".
97
+ // Keep this explicit so generic "rate limit exhausted, retry..." throttles remain retryable.
96
98
  const USAGE_LIMIT_PATTERN =
97
- /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|model.?limit|model_limit_reached|message.?limit|message_limit_reached|limit for this model|quota.?exceeded|out_of_credits|request would exceed your account.?s rate limit|resource has been exhausted[^\n]*(?:quota|limit)/i;
99
+ /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|model.?limit|model_limit_reached|message.?limit|message_limit_reached|limit for this model|weekly\/monthly\s+limit\s+exhausted|quota.?exceeded|out_of_credits|request would exceed your account.?s rate limit|resource has been exhausted[^\n]*(?:quota|limit)/i;
98
100
  export function isUsageLimitError(errorMessage: string): boolean {
99
101
  return USAGE_LIMIT_PATTERN.test(errorMessage);
100
102
  }
package/src/stream.ts CHANGED
@@ -98,6 +98,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
98
98
  "opencode-zen": "OPENCODE_API_KEY",
99
99
  cursor: "CURSOR_ACCESS_TOKEN",
100
100
  deepseek: "DEEPSEEK_API_KEY",
101
+ deepinfra: "DEEPINFRA_API_KEY",
101
102
  "openai-codex": "OPENAI_CODEX_OAUTH_TOKEN",
102
103
  "azure-openai": "AZURE_OPENAI_API_KEY",
103
104
  "azure-openai-responses": "AZURE_OPENAI_API_KEY",
package/src/types.ts CHANGED
@@ -116,6 +116,7 @@ export type KnownProvider =
116
116
  | "gitlab-duo"
117
117
  | "cursor"
118
118
  | "deepseek"
119
+ | "deepinfra"
119
120
  | "xai"
120
121
  | "groq"
121
122
  | "cerebras"
@@ -224,17 +225,20 @@ export function resolveServiceTier(
224
225
 
225
226
  /**
226
227
  * True when the (possibly scoped) tier should be sent as OpenAI's
227
- * `service_tier` request field for the given provider. Non-OpenAI
228
- * providers, unsupported tiers (`"auto"`, `"default"`), and scope
229
- * mismatches all return false.
228
+ * `service_tier` request field for the given provider. OpenAI accepts
229
+ * `flex`, `scale`, and `priority`; DeepInfra accepts `priority`.
230
+ * Unsupported tiers (`"auto"`, `"default"`) and scope mismatches return false.
230
231
  */
231
232
  export function shouldSendServiceTier(
232
233
  serviceTier: ServiceTier | null | undefined,
233
234
  provider: Provider | undefined,
234
235
  ): boolean {
235
- if (provider !== "openai" && provider !== "openai-codex") return false;
236
236
  const resolved = resolveServiceTier(serviceTier, provider);
237
- return resolved === "flex" || resolved === "scale" || resolved === "priority";
237
+ if (provider === "openai" || provider === "openai-codex") {
238
+ return resolved === "flex" || resolved === "scale" || resolved === "priority";
239
+ }
240
+ if (provider === "deepinfra") return resolved === "priority";
241
+ return false;
238
242
  }
239
243
 
240
244
  /**
@@ -252,7 +256,9 @@ export function getPriorityPremiumRequests(
252
256
  if (resolveServiceTier(serviceTier, provider) !== "priority") return 0;
253
257
  // Only providers that realize `priority` on the wire bill the user.
254
258
  // Everywhere else, the field is silently dropped and nothing is charged.
255
- return provider === "openai" || provider === "openai-codex" || provider === "anthropic" ? 1 : 0;
259
+ return provider === "openai" || provider === "openai-codex" || provider === "anthropic" || provider === "deepinfra"
260
+ ? 1
261
+ : 0;
256
262
  }
257
263
 
258
264
  export interface ProviderSessionState {
@@ -2,11 +2,11 @@ import * as http2 from "node:http2";
2
2
  import { create, fromBinary, toBinary } from "@bufbuild/protobuf";
3
3
  import * as z from "zod/v4";
4
4
  import { getBundledModels } from "../../models";
5
+ import { CURSOR_CLIENT_VERSION } from "../../providers/cursor/client-version";
5
6
  import { GetUsableModelsRequestSchema, GetUsableModelsResponseSchema } from "../../providers/cursor/gen/agent_pb";
6
7
  import type { Model } from "../../types";
7
8
 
8
9
  const CURSOR_DEFAULT_BASE_URL = "https://api2.cursor.sh";
9
- const CURSOR_DEFAULT_CLIENT_VERSION = "cli-2026.02.13-41ac335";
10
10
  const CURSOR_GET_USABLE_MODELS_PATH = "/agent.v1.AgentService/GetUsableModels";
11
11
 
12
12
  const DEFAULT_CONTEXT_WINDOW = 200_000;
@@ -91,7 +91,7 @@ function buildRequestHeaders(options: CursorModelDiscoveryOptions): Record<strin
91
91
  te: "trailers",
92
92
  authorization: `Bearer ${options.apiKey}`,
93
93
  "x-ghost-mode": "true",
94
- "x-cursor-client-version": options.clientVersion ?? CURSOR_DEFAULT_CLIENT_VERSION,
94
+ "x-cursor-client-version": options.clientVersion ?? CURSOR_CLIENT_VERSION,
95
95
  "x-cursor-client-type": "cli",
96
96
  };
97
97
  }
@@ -3,10 +3,10 @@ import { createApiKeyLogin } from "./api-key-login";
3
3
 
4
4
  export const loginFugu = createApiKeyLogin({
5
5
  providerLabel: "Sakana Fugu",
6
- authUrl: "https://fugu.sakana.ai/",
6
+ authUrl: "https://console.sakana.ai/api-keys",
7
7
  instructions: "Create or copy your Sakana Fugu API key",
8
8
  promptMessage: "Paste your Sakana Fugu API key",
9
- placeholder: "fugu_...",
9
+ placeholder: "fish_...",
10
10
  validation: {
11
11
  kind: "models-endpoint",
12
12
  provider: "Sakana Fugu",
@@ -60,6 +60,11 @@ const builtInOAuthProviders: OAuthProviderInfo[] = [
60
60
  name: "DeepSeek",
61
61
  available: true,
62
62
  },
63
+ {
64
+ id: "deepinfra",
65
+ name: "DeepInfra",
66
+ available: true,
67
+ },
63
68
  {
64
69
  id: "xai",
65
70
  name: "xAI",
@@ -359,6 +364,7 @@ export async function refreshOAuthToken(
359
364
  case "fireworks":
360
365
  case "firepass":
361
366
  case "fugu":
367
+ case "deepinfra":
362
368
  case "nvidia":
363
369
  case "nanogpt":
364
370
  case "synthetic":
@@ -15,6 +15,7 @@ export type OAuthProvider =
15
15
  | "cloudflare-ai-gateway"
16
16
  | "cursor"
17
17
  | "deepseek"
18
+ | "deepinfra"
18
19
  | "fireworks"
19
20
  | "firepass"
20
21
  | "fugu"