@sayknow-cli/ai 0.3.8 → 0.3.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -13,6 +13,7 @@ import {
13
13
  anthropicModelManagerOptions,
14
14
  cerebrasModelManagerOptions,
15
15
  cloudflareAiGatewayModelManagerOptions,
16
+ deepinfraModelManagerOptions,
16
17
  deepseekModelManagerOptions,
17
18
  firepassModelManagerOptions,
18
19
  fireworksModelManagerOptions,
@@ -168,6 +169,12 @@ export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [
168
169
  config => deepseekModelManagerOptions(config),
169
170
  catalog("DeepSeek", ["DEEPSEEK_API_KEY"]),
170
171
  ),
172
+ catalogDescriptor(
173
+ "deepinfra",
174
+ "deepseek-ai/DeepSeek-V3.2",
175
+ config => deepinfraModelManagerOptions(config),
176
+ catalog("DeepInfra", ["DEEPINFRA_API_KEY"]),
177
+ ),
171
178
  descriptor("mistral", "devstral-medium-latest", config => mistralModelManagerOptions(config)),
172
179
  catalogDescriptor(
173
180
  "nvidia",
@@ -665,6 +665,17 @@ export function deepseekModelManagerOptions(
665
665
  ): ModelManagerOptions<"openai-completions"> {
666
666
  return createSimpleOpenAICompletionsOptions("deepseek", "https://api.deepseek.com", config);
667
667
  }
668
+
669
+ export interface DeepInfraModelManagerConfig {
670
+ apiKey?: string;
671
+ baseUrl?: string;
672
+ }
673
+
674
+ export function deepinfraModelManagerOptions(
675
+ config?: DeepInfraModelManagerConfig,
676
+ ): ModelManagerOptions<"openai-completions"> {
677
+ return createSimpleOpenAICompletionsOptions("deepinfra", "https://api.deepinfra.com/v1/openai", config);
678
+ }
668
679
  // ---------------------------------------------------------------------------
669
680
  // 7.5 Fireworks
670
681
  // ---------------------------------------------------------------------------
@@ -2282,6 +2293,8 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
2282
2293
  requiresAssistantContentForToolCalls: true,
2283
2294
  },
2284
2295
  }),
2296
+ // --- DeepInfra ---
2297
+ openAiCompletionsDescriptor("deepinfra", "deepinfra", "https://api.deepinfra.com/v1/openai"),
2285
2298
  ];
2286
2299
 
2287
2300
  const MODELS_DEV_PROVIDER_DESCRIPTORS_CODING_PLANS: readonly ModelsDevProviderDescriptor[] = [
@@ -0,0 +1,10 @@
1
+ /**
2
+ * The Cursor client version reported to api2.cursor.sh.
3
+ *
4
+ * Every call against the Cursor backend must send the same
5
+ * x-cursor-client-version: the backend gates features and minimum versions on
6
+ * it, so a drift between the agent Run path and model discovery makes one of
7
+ * them fail while the other keeps working. Keep this as the single source of
8
+ * truth for the header value.
9
+ */
10
+ export const CURSOR_CLIENT_VERSION = "cli-2026.02.13-41ac335";
@@ -31,6 +31,7 @@ import { parseStreamingJson } from "../utils/json-parse";
31
31
  import { formatErrorMessageWithRetryAfter } from "../utils/retry-after";
32
32
  import { flattenToolRootCombinators, toolWireSchema } from "../utils/schema";
33
33
  import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
34
+ import { CURSOR_CLIENT_VERSION } from "./cursor/client-version";
34
35
  import type { McpToolDefinition } from "./cursor/gen/agent_pb";
35
36
  import {
36
37
  AgentClientMessageSchema,
@@ -131,7 +132,7 @@ import {
131
132
  } from "./cursor/gen/agent_pb";
132
133
 
133
134
  export const CURSOR_API_URL = "https://api2.cursor.sh";
134
- export const CURSOR_CLIENT_VERSION = "cli-2026.01.09-231024f";
135
+ export { CURSOR_CLIENT_VERSION };
135
136
 
136
137
  const conversationStateCache = new Map<string, ConversationStateStructure>();
137
138
  const conversationBlobStores = new Map<string, Map<string, Uint8Array>>();
@@ -4,7 +4,7 @@
4
4
  * GeminiCLI/VERSION/MODEL (PLATFORM; ARCH; SURFACE)
5
5
  */
6
6
  export function getGeminiCliUserAgent(modelId = "gemini-3.1-pro-preview"): string {
7
- const version = process.env.PI_AI_GEMINI_CLI_VERSION || "0.46.0";
7
+ const version = process.env.SKC_AI_GEMINI_CLI_VERSION || process.env.PI_AI_GEMINI_CLI_VERSION || "0.49.0";
8
8
  const platform = process.platform === "win32" ? "win32" : process.platform;
9
9
  const arch = process.arch === "x64" ? "x64" : process.arch;
10
10
  return `GeminiCLI/${version}/${modelId} (${platform}; ${arch}; terminal)`;
@@ -35,7 +35,7 @@ import {
35
35
  type ToolChoice,
36
36
  type ToolResultMessage,
37
37
  } from "../types";
38
- import { normalizeSystemPrompts } from "../utils";
38
+ import { normalizeSystemPrompts, sanitizeJsonStrings } from "../utils";
39
39
  import { createAbortSourceTracker } from "../utils/abort";
40
40
  import { AssistantMessageEventStream } from "../utils/event-stream";
41
41
  import { toFirepassWireModelId, toFireworksWireModelId } from "../utils/fireworks-model-id";
@@ -166,7 +166,7 @@ function normalizeStreamingContentText(content: unknown): string {
166
166
  function serializeToolArguments(value: unknown): string {
167
167
  if (value && typeof value === "object" && !Array.isArray(value)) {
168
168
  try {
169
- return JSON.stringify(value);
169
+ return JSON.stringify(sanitizeJsonStrings(value));
170
170
  } catch {
171
171
  return "{}";
172
172
  }
@@ -178,7 +178,7 @@ function serializeToolArguments(value: unknown): string {
178
178
  try {
179
179
  const parsed = JSON.parse(trimmed);
180
180
  if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) {
181
- return JSON.stringify(parsed);
181
+ return JSON.stringify(sanitizeJsonStrings(parsed));
182
182
  }
183
183
  } catch {}
184
184
  return "{}";
@@ -13,7 +13,7 @@ export type RateLimitReason =
13
13
  const QUOTA_EXHAUSTED_BACKOFF_MS = 30 * 60 * 1000; // 30 min
14
14
  const RATE_LIMIT_EXCEEDED_BACKOFF_MS = 30 * 1000; // 30s
15
15
  const MODEL_CAPACITY_BASE_MS = 45 * 1000; // 45s base
16
- const MODEL_CAPACITY_JITTER_MS = 30 * 1000; // ±15s
16
+ const MODEL_CAPACITY_JITTER_MS = 30 * 1000; // uniform +0–30s above base → 45–75s total
17
17
  const SERVER_ERROR_BACKOFF_MS = 20 * 1000; // 20s
18
18
 
19
19
  /**
@@ -93,8 +93,10 @@ export function calculateRateLimitBackoffMs(reason: RateLimitReason): number {
93
93
  }
94
94
 
95
95
  /** Detect usage/quota limit errors in error messages (persistent, requires credential switch). */
96
+ // ZAI reports durable token exhaustion as "[1310][Weekly/Monthly Limit Exhausted...]".
97
+ // Keep this explicit so generic "rate limit exhausted, retry..." throttles remain retryable.
96
98
  const USAGE_LIMIT_PATTERN =
97
- /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|model.?limit|model_limit_reached|message.?limit|message_limit_reached|limit for this model|quota.?exceeded|out_of_credits|request would exceed your account.?s rate limit|resource has been exhausted[^\n]*(?:quota|limit)/i;
99
+ /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|model.?limit|model_limit_reached|message.?limit|message_limit_reached|limit for this model|weekly\/monthly\s+limit\s+exhausted|quota.?exceeded|out_of_credits|request would exceed your account.?s rate limit|resource has been exhausted[^\n]*(?:quota|limit)/i;
98
100
  export function isUsageLimitError(errorMessage: string): boolean {
99
101
  return USAGE_LIMIT_PATTERN.test(errorMessage);
100
102
  }
package/src/stream.ts CHANGED
@@ -98,6 +98,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
98
98
  "opencode-zen": "OPENCODE_API_KEY",
99
99
  cursor: "CURSOR_ACCESS_TOKEN",
100
100
  deepseek: "DEEPSEEK_API_KEY",
101
+ deepinfra: "DEEPINFRA_API_KEY",
101
102
  "openai-codex": "OPENAI_CODEX_OAUTH_TOKEN",
102
103
  "azure-openai": "AZURE_OPENAI_API_KEY",
103
104
  "azure-openai-responses": "AZURE_OPENAI_API_KEY",
package/src/types.ts CHANGED
@@ -116,6 +116,7 @@ export type KnownProvider =
116
116
  | "gitlab-duo"
117
117
  | "cursor"
118
118
  | "deepseek"
119
+ | "deepinfra"
119
120
  | "xai"
120
121
  | "groq"
121
122
  | "cerebras"
@@ -186,7 +187,7 @@ export type CacheRetention = "none" | "short" | "long";
186
187
  *
187
188
  * The unscoped values (`"auto"`, `"default"`, `"flex"`, `"scale"`,
188
189
  * `"priority"`) are passed through to providers that understand them
189
- * (OpenAI's `service_tier` field directly; Anthropic translates
190
+ * (OpenAI and DeepInfra's `service_tier` field directly; Anthropic translates
190
191
  * `"priority"` into `speed: "fast"` on supported Opus models).
191
192
  *
192
193
  * The scoped values target a specific provider family and behave as the
@@ -223,17 +224,17 @@ export function resolveServiceTier(
223
224
  }
224
225
 
225
226
  /**
226
- * True when the (possibly scoped) tier should be sent as OpenAI's
227
- * `service_tier` request field for the given provider. Non-OpenAI
228
- * providers, unsupported tiers (`"auto"`, `"default"`), and scope
229
- * mismatches all return false.
227
+ * True when the (possibly scoped) tier should be sent as an OpenAI-compatible
228
+ * `service_tier` request field for providers that support it. Unsupported tiers
229
+ * (`"auto"`, `"default"`) and scope mismatches all return false.
230
230
  */
231
231
  export function shouldSendServiceTier(
232
232
  serviceTier: ServiceTier | null | undefined,
233
233
  provider: Provider | undefined,
234
234
  ): boolean {
235
- if (provider !== "openai" && provider !== "openai-codex") return false;
236
235
  const resolved = resolveServiceTier(serviceTier, provider);
236
+ if (provider === "deepinfra") return resolved === "priority";
237
+ if (provider !== "openai" && provider !== "openai-codex") return false;
237
238
  return resolved === "flex" || resolved === "scale" || resolved === "priority";
238
239
  }
239
240
 
@@ -252,7 +253,9 @@ export function getPriorityPremiumRequests(
252
253
  if (resolveServiceTier(serviceTier, provider) !== "priority") return 0;
253
254
  // Only providers that realize `priority` on the wire bill the user.
254
255
  // Everywhere else, the field is silently dropped and nothing is charged.
255
- return provider === "openai" || provider === "openai-codex" || provider === "anthropic" ? 1 : 0;
256
+ return provider === "openai" || provider === "openai-codex" || provider === "anthropic" || provider === "deepinfra"
257
+ ? 1
258
+ : 0;
256
259
  }
257
260
 
258
261
  export interface ProviderSessionState {
@@ -2,11 +2,11 @@ import * as http2 from "node:http2";
2
2
  import { create, fromBinary, toBinary } from "@bufbuild/protobuf";
3
3
  import * as z from "zod/v4";
4
4
  import { getBundledModels } from "../../models";
5
+ import { CURSOR_CLIENT_VERSION } from "../../providers/cursor/client-version";
5
6
  import { GetUsableModelsRequestSchema, GetUsableModelsResponseSchema } from "../../providers/cursor/gen/agent_pb";
6
7
  import type { Model } from "../../types";
7
8
 
8
9
  const CURSOR_DEFAULT_BASE_URL = "https://api2.cursor.sh";
9
- const CURSOR_DEFAULT_CLIENT_VERSION = "cli-2026.02.13-41ac335";
10
10
  const CURSOR_GET_USABLE_MODELS_PATH = "/agent.v1.AgentService/GetUsableModels";
11
11
 
12
12
  const DEFAULT_CONTEXT_WINDOW = 200_000;
@@ -91,7 +91,7 @@ function buildRequestHeaders(options: CursorModelDiscoveryOptions): Record<strin
91
91
  te: "trailers",
92
92
  authorization: `Bearer ${options.apiKey}`,
93
93
  "x-ghost-mode": "true",
94
- "x-cursor-client-version": options.clientVersion ?? CURSOR_DEFAULT_CLIENT_VERSION,
94
+ "x-cursor-client-version": options.clientVersion ?? CURSOR_CLIENT_VERSION,
95
95
  "x-cursor-client-type": "cli",
96
96
  };
97
97
  }
@@ -3,10 +3,10 @@ import { createApiKeyLogin } from "./api-key-login";
3
3
 
4
4
  export const loginFugu = createApiKeyLogin({
5
5
  providerLabel: "Sakana Fugu",
6
- authUrl: "https://fugu.sakana.ai/",
6
+ authUrl: "https://console.sakana.ai/api-keys",
7
7
  instructions: "Create or copy your Sakana Fugu API key",
8
8
  promptMessage: "Paste your Sakana Fugu API key",
9
- placeholder: "fugu_...",
9
+ placeholder: "fish_...",
10
10
  validation: {
11
11
  kind: "models-endpoint",
12
12
  provider: "Sakana Fugu",
@@ -60,6 +60,11 @@ const builtInOAuthProviders: OAuthProviderInfo[] = [
60
60
  name: "DeepSeek",
61
61
  available: true,
62
62
  },
63
+ {
64
+ id: "deepinfra",
65
+ name: "DeepInfra",
66
+ available: true,
67
+ },
63
68
  {
64
69
  id: "xai",
65
70
  name: "xAI",
@@ -15,6 +15,7 @@ export type OAuthProvider =
15
15
  | "cloudflare-ai-gateway"
16
16
  | "cursor"
17
17
  | "deepseek"
18
+ | "deepinfra"
18
19
  | "fireworks"
19
20
  | "firepass"
20
21
  | "fugu"