@sayknow-cli/ai 0.3.7 → 0.3.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -70,6 +70,7 @@ export declare class RemoteAuthCredentialStore implements AuthCredentialStore {
70
70
  getCache(key: string): string | null;
71
71
  setCache(key: string, value: string, expiresAtSec: number): void;
72
72
  cleanExpiredCache(): void;
73
+ deleteCachePrefix(prefix: string): void;
73
74
  /**
74
75
  * Store-level hook consumed by `AuthStorage` — routes refresh through the
75
76
  * broker so the actual refresh token never leaves the broker host. Returns
@@ -420,6 +420,7 @@ export declare const credentialIfAbsentUploadResponseSchema: z.ZodObject<{
420
420
  "skipped-existing-fallback": "skipped-existing-fallback";
421
421
  "skipped-existing-runtime": "skipped-existing-runtime";
422
422
  "skipped-invalid": "skipped-invalid";
423
+ "updated-existing": "updated-existing";
423
424
  }>;
424
425
  entries: z.ZodArray<z.ZodObject<{
425
426
  id: z.ZodNumber;
@@ -102,7 +102,7 @@ export interface AuthCredentialSnapshotEntry {
102
102
  credential: SnapshotCredential;
103
103
  identityKey: string | null;
104
104
  }
105
- export type AuthCredentialIfAbsentReason = "inserted" | "skipped-existing" | "skipped-existing-runtime" | "skipped-existing-config" | "skipped-existing-env" | "skipped-existing-fallback" | "skipped-invalid";
105
+ export type AuthCredentialIfAbsentReason = "inserted" | "updated-existing" | "skipped-existing" | "skipped-existing-runtime" | "skipped-existing-config" | "skipped-existing-env" | "skipped-existing-fallback" | "skipped-invalid";
106
106
  export interface AuthCredentialIfAbsentResult {
107
107
  inserted: boolean;
108
108
  reason: AuthCredentialIfAbsentReason;
@@ -147,6 +147,7 @@ export interface AuthCredentialStore {
147
147
  includeExpired?: boolean;
148
148
  }): string | null;
149
149
  setCache(key: string, value: string, expiresAtSec: number): void;
150
+ deleteCachePrefix?(prefix: string): void;
150
151
  cleanExpiredCache(): void;
151
152
  /**
152
153
  * Optional store-supplied OAuth refresh. When present, `AuthStorage` uses
@@ -646,6 +647,7 @@ export declare class SqliteAuthCredentialStore implements AuthCredentialStore {
646
647
  includeExpired?: boolean;
647
648
  }): string | null;
648
649
  setCache(key: string, value: string, expiresAtSec: number): void;
650
+ deleteCachePrefix(prefix: string): void;
649
651
  cleanExpiredCache(): void;
650
652
  /**
651
653
  * Save OAuth credentials for a provider.
@@ -58,6 +58,11 @@ export interface DeepSeekModelManagerConfig {
58
58
  baseUrl?: string;
59
59
  }
60
60
  export declare function deepseekModelManagerOptions(config?: DeepSeekModelManagerConfig): ModelManagerOptions<"openai-completions">;
61
+ export interface DeepInfraModelManagerConfig {
62
+ apiKey?: string;
63
+ baseUrl?: string;
64
+ }
65
+ export declare function deepinfraModelManagerOptions(config?: DeepInfraModelManagerConfig): ModelManagerOptions<"openai-completions">;
61
66
  export interface FireworksModelManagerConfig {
62
67
  apiKey?: string;
63
68
  baseUrl?: string;
@@ -0,0 +1,10 @@
1
+ /**
2
+ * The Cursor client version reported to api2.cursor.sh.
3
+ *
4
+ * Every call against the Cursor backend must send the same
5
+ * x-cursor-client-version: the backend gates features and minimum versions on
6
+ * it, so a drift between the agent Run path and model discovery makes one of
7
+ * them fail while the other keeps working. Keep this as the single source of
8
+ * truth for the header value.
9
+ */
10
+ export declare const CURSOR_CLIENT_VERSION = "cli-2026.02.13-41ac335";
@@ -1,7 +1,8 @@
1
1
  import { type JsonValue } from "@bufbuild/protobuf";
2
2
  import type { CursorExecHandlerResult, CursorExecHandlers, CursorToolResultHandler, Message, StreamFunction, StreamOptions, ToolResultMessage } from "../types";
3
+ import { CURSOR_CLIENT_VERSION } from "./cursor/client-version";
3
4
  export declare const CURSOR_API_URL = "https://api2.cursor.sh";
4
- export declare const CURSOR_CLIENT_VERSION = "cli-2026.01.09-231024f";
5
+ export { CURSOR_CLIENT_VERSION };
5
6
  /** Drop all cached state + blob bytes for a conversation (F15 bound + session-teardown hook). */
6
7
  export declare function disposeCursorConversation(conversationId: string): void;
7
8
  export interface CursorOptions extends StreamOptions {
@@ -0,0 +1,3 @@
1
+ import type { FetchImpl } from "../types";
2
+ export declare function isOpenAIUsageExhaustionResponse(bodyText: string, retryAfterMs: number | undefined, retryDelayCapMs: number): boolean;
3
+ export declare function wrapOpenAIFetchForBoundedRateLimits(baseFetch: FetchImpl, maxRetryDelayMs: number | undefined): FetchImpl;
@@ -48,7 +48,7 @@ export interface ThinkingConfig {
48
48
  /** Provider-specific transport used to encode the selected effort. */
49
49
  mode: ThinkingControlMode;
50
50
  }
51
- export type KnownProvider = "alibaba-coding-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
51
+ export type KnownProvider = "alibaba-coding-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
52
52
  export type Provider = KnownProvider | string;
53
53
  import type { Effort } from "./model-thinking";
54
54
  /** Token budgets for each thinking level (token-based providers only) */
@@ -106,9 +106,9 @@ export type ResolvedServiceTier = Exclude<ServiceTier, "openai-only" | "claude-o
106
106
  export declare function resolveServiceTier(serviceTier: ServiceTier | null | undefined, provider: Provider | undefined): ResolvedServiceTier | undefined;
107
107
  /**
108
108
  * True when the (possibly scoped) tier should be sent as OpenAI's
109
- * `service_tier` request field for the given provider. Non-OpenAI
110
- * providers, unsupported tiers (`"auto"`, `"default"`), and scope
111
- * mismatches all return false.
109
+ * `service_tier` request field for the given provider. OpenAI accepts
110
+ * `flex`, `scale`, and `priority`; DeepInfra accepts `priority`.
111
+ * Unsupported tiers (`"auto"`, `"default"`) and scope mismatches return false.
112
112
  */
113
113
  export declare function shouldSendServiceTier(serviceTier: ServiceTier | null | undefined, provider: Provider | undefined): boolean;
114
114
  /**
@@ -7,7 +7,7 @@ export type OAuthCredentials = {
7
7
  email?: string;
8
8
  accountId?: string;
9
9
  };
10
- export type OAuthProvider = "alibaba-coding-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
10
+ export type OAuthProvider = "alibaba-coding-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
11
11
  export type OAuthProviderId = OAuthProvider | (string & {});
12
12
  export type OAuthPrompt = {
13
13
  message: string;
package/package.json CHANGED
@@ -1,9 +1,9 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@sayknow-cli/ai",
4
- "version": "0.3.7",
4
+ "version": "0.3.9",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
- "homepage": "https://github.com/jaybeyond/Sayknow_CLI",
6
+ "homepage": "https://sayknow-cli.com",
7
7
  "author": "jaybeyond",
8
8
  "contributors": [
9
9
  "Mario Zechner"
@@ -43,7 +43,7 @@
43
43
  "dependencies": {
44
44
  "@anthropic-ai/sdk": "^0.94.0",
45
45
  "@bufbuild/protobuf": "^2.12.0",
46
- "@sayknow-cli/utils": "0.3.7",
46
+ "@sayknow-cli/utils": "0.3.9",
47
47
  "openai": "^6.36.0",
48
48
  "partial-json": "^0.1.7",
49
49
  "zod": "4.4.3"
@@ -86,6 +86,7 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
86
86
  #cache: Map<string, CacheEntry> = new Map();
87
87
  #usageCache?: UsageCacheEntry;
88
88
  #usageInflight?: Promise<UsageReport[] | null>;
89
+ #usageCacheEpoch = 0;
89
90
  #closed = false;
90
91
  /**
91
92
  * `true` once the SSE consumer received its first frame and hasn't dropped
@@ -465,6 +466,19 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
465
466
  }
466
467
  }
467
468
 
469
+ deleteCachePrefix(prefix: string): void {
470
+ for (const key of this.#cache.keys()) {
471
+ if (key.startsWith(prefix)) this.#cache.delete(key);
472
+ }
473
+ if (prefix.startsWith("usage_cache:")) this.#invalidateUsageCache();
474
+ }
475
+
476
+ #invalidateUsageCache(): void {
477
+ this.#usageCache = undefined;
478
+ this.#usageInflight = undefined;
479
+ this.#usageCacheEpoch += 1;
480
+ }
481
+
468
482
  /**
469
483
  * Store-level hook consumed by `AuthStorage` — routes refresh through the
470
484
  * broker so the actual refresh token never leaves the broker host. Returns
@@ -561,10 +575,11 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
561
575
  return Promise.resolve(cached.reports);
562
576
  }
563
577
  if (this.#usageInflight) return this.#usageInflight;
578
+ const epoch = this.#usageCacheEpoch;
564
579
  const inflight = this.#client
565
580
  .fetchUsage()
566
581
  .then(body => {
567
- this.#usageCache = { reports: body.reports, fetchedAt: Date.now() };
582
+ if (this.#usageCacheEpoch === epoch) this.#usageCache = { reports: body.reports, fetchedAt: Date.now() };
568
583
  return body.reports;
569
584
  })
570
585
  .catch(error => {
@@ -572,7 +587,7 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
572
587
  return null;
573
588
  })
574
589
  .finally(() => {
575
- this.#usageInflight = undefined;
590
+ if (this.#usageCacheEpoch === epoch) this.#usageInflight = undefined;
576
591
  });
577
592
  this.#usageInflight = inflight;
578
593
  return inflight;
@@ -187,6 +187,7 @@ export const credentialDisableResponseSchema = z
187
187
  // ─── Upload ────────────────────────────────────────────────────────────────
188
188
  const credentialIfAbsentReasonSchema = z.enum([
189
189
  "inserted",
190
+ "updated-existing",
190
191
  "skipped-existing",
191
192
  "skipped-existing-runtime",
192
193
  "skipped-existing-config",
@@ -32,6 +32,7 @@ import { kimiUsageProvider } from "./usage/kimi";
32
32
  import { codexRankingStrategy, openaiCodexUsageProvider } from "./usage/openai-codex";
33
33
  import { zaiUsageProvider } from "./usage/zai";
34
34
  import { getOAuthApiKey, getOAuthProvider, refreshOAuthToken, resolveOAuthStorageProvider } from "./utils/oauth";
35
+ import { loginDeepInfra } from "./utils/oauth/deepinfra";
35
36
  import { loginDeepSeek } from "./utils/oauth/deepseek";
36
37
  import { loginOpenAICodexDevice } from "./utils/oauth/openai-codex";
37
38
  import type { OAuthController, OAuthCredentials, OAuthProvider, OAuthProviderId } from "./utils/oauth/types";
@@ -153,6 +154,7 @@ export interface AuthCredentialSnapshotEntry {
153
154
 
154
155
  export type AuthCredentialIfAbsentReason =
155
156
  | "inserted"
157
+ | "updated-existing"
156
158
  | "skipped-existing"
157
159
  | "skipped-existing-runtime"
158
160
  | "skipped-existing-config"
@@ -209,6 +211,7 @@ export interface AuthCredentialStore {
209
211
  deleteAuthCredentialsForProvider(provider: string, disabledCause: string): void;
210
212
  getCache(key: string, options?: { includeExpired?: boolean }): string | null;
211
213
  setCache(key: string, value: string, expiresAtSec: number): void;
214
+ deleteCachePrefix?(prefix: string): void;
212
215
  cleanExpiredCache(): void;
213
216
  /**
214
217
  * Optional store-supplied OAuth refresh. When present, `AuthStorage` uses
@@ -471,6 +474,7 @@ interface UsageCache {
471
474
  get<T>(key: string): UsageCacheEntry<T> | undefined;
472
475
  getStale<T>(key: string): UsageCacheEntry<T> | undefined;
473
476
  set<T>(key: string, entry: UsageCacheEntry<T>): void;
477
+ deletePrefix?(prefix: string): void;
474
478
  cleanup?(): void;
475
479
  }
476
480
 
@@ -661,6 +665,10 @@ class AuthStorageUsageCache implements UsageCache {
661
665
  this.store.setCache(`${USAGE_CACHE_PREFIX}${key}`, payload, Math.floor(durableExpiresAt / 1000));
662
666
  }
663
667
 
668
+ deletePrefix(prefix: string): void {
669
+ this.store.deleteCachePrefix?.(`${USAGE_CACHE_PREFIX}${prefix}`);
670
+ }
671
+
664
672
  cleanup(): void {
665
673
  this.store.cleanExpiredCache();
666
674
  }
@@ -1296,8 +1304,6 @@ export class AuthStorage {
1296
1304
  return this.#snapshotSkipResult(storageProvider, "skipped-existing-runtime");
1297
1305
  if (this.#configOverrides.has(storageProvider))
1298
1306
  return this.#snapshotSkipResult(storageProvider, "skipped-existing-config");
1299
- if (this.#getCredentialsForProvider(storageProvider).length > 0)
1300
- return this.#snapshotSkipResult(storageProvider, "skipped-existing");
1301
1307
  if (getEnvApiKey(storageProvider)) return this.#snapshotSkipResult(storageProvider, "skipped-existing-env");
1302
1308
  if (this.#fallbackResolver?.(storageProvider))
1303
1309
  return this.#snapshotSkipResult(storageProvider, "skipped-existing-fallback");
@@ -1310,6 +1316,7 @@ export class AuthStorage {
1310
1316
  result.entries.map(entry => ({ id: entry.id, credential: entry.credential })),
1311
1317
  );
1312
1318
  this.#resetProviderAssignments(storageProvider);
1319
+ if (result.inserted) this.#invalidateUsageCacheForProvider(storageProvider);
1313
1320
  return {
1314
1321
  inserted: result.inserted,
1315
1322
  reason: result.reason,
@@ -1327,6 +1334,13 @@ export class AuthStorage {
1327
1334
  stored.map(record => ({ id: record.id, credential: record.credential })),
1328
1335
  );
1329
1336
  this.#resetProviderAssignments(provider);
1337
+ this.#invalidateUsageCacheForProvider(provider);
1338
+ }
1339
+
1340
+ #invalidateUsageCacheForProvider(provider: string): void {
1341
+ this.#usageRequestInFlight.clear();
1342
+ this.#usageReportsInFlight.clear();
1343
+ this.#usageCache.deletePrefix?.(`report:${provider}:`);
1330
1344
  }
1331
1345
 
1332
1346
  /**
@@ -1593,6 +1607,11 @@ export class AuthStorage {
1593
1607
  await saveApiKeyCredential(apiKey);
1594
1608
  return;
1595
1609
  }
1610
+ case "deepinfra": {
1611
+ const apiKey = await loginDeepInfra(ctrl);
1612
+ await saveApiKeyCredential(apiKey);
1613
+ return;
1614
+ }
1596
1615
  case "xai": {
1597
1616
  const { loginXai } = await import("./utils/oauth/xai");
1598
1617
  credentials = await loginXai({
@@ -3699,6 +3718,7 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
3699
3718
  #getCacheStmt: Statement;
3700
3719
  #getCacheIncludingExpiredStmt: Statement;
3701
3720
  #upsertCacheStmt: Statement;
3721
+ #deleteCachePrefixStmt: Statement;
3702
3722
  #deleteExpiredCacheStmt: Statement;
3703
3723
  #closed = false;
3704
3724
 
@@ -3738,6 +3758,7 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
3738
3758
  this.#upsertCacheStmt = this.#db.prepare(
3739
3759
  "INSERT INTO cache (key, value, expires_at) VALUES (?, ?, ?) ON CONFLICT(key) DO UPDATE SET value = excluded.value, expires_at = excluded.expires_at",
3740
3760
  );
3761
+ this.#deleteCachePrefixStmt = this.#db.prepare("DELETE FROM cache WHERE substr(key, 1, ?) = ?");
3741
3762
  this.#deleteExpiredCacheStmt = this.#db.prepare(`DELETE FROM cache WHERE expires_at <= ${SQLITE_NOW_EPOCH}`);
3742
3763
  }
3743
3764
 
@@ -4096,16 +4117,58 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
4096
4117
  }
4097
4118
 
4098
4119
  const writeIfAbsent = this.#db.transaction(
4099
- (providerName: string, record: SerializedCredentialRecord): AuthCredentialIfAbsentResult => {
4120
+ (
4121
+ providerName: string,
4122
+ item: AuthCredential,
4123
+ record: SerializedCredentialRecord,
4124
+ ): AuthCredentialIfAbsentResult => {
4100
4125
  const existingRows = this.#listActiveByProviderStmt.all(providerName) as AuthRow[];
4101
- const existing: StoredAuthCredential[] = [];
4126
+ const existing: Array<{
4127
+ id: number;
4128
+ credential: AuthCredential;
4129
+ identityKey: string | null;
4130
+ }> = [];
4102
4131
  for (const row of existingRows) {
4103
4132
  const activeCredential = deserializeCredential(row);
4104
4133
  if (!activeCredential) continue;
4105
- existing.push(toStoredAuthCredential(row, activeCredential));
4134
+ existing.push({
4135
+ id: row.id,
4136
+ credential: activeCredential,
4137
+ identityKey: resolveRowCredentialIdentityKey(providerName, row),
4138
+ });
4106
4139
  }
4107
4140
  if (existing.length > 0) {
4108
- return { inserted: false, reason: "skipped-existing", provider: providerName, entries: existing };
4141
+ let targetId: number | null = null;
4142
+ for (const row of existing) {
4143
+ if (!matchesReplacementCredential(providerName, row.credential, row.identityKey, item)) continue;
4144
+ if (targetId === null) {
4145
+ targetId = row.id;
4146
+ this.#updateStmt.run(record.credentialType, record.data, record.identityKey, row.id);
4147
+ } else {
4148
+ this.#deleteStmt.run("replaced by newer credential", row.id);
4149
+ }
4150
+ }
4151
+
4152
+ if (targetId !== null) {
4153
+ return {
4154
+ inserted: true,
4155
+ reason: "updated-existing",
4156
+ provider: providerName,
4157
+ entries: this.listAuthCredentials(providerName),
4158
+ };
4159
+ }
4160
+
4161
+ return {
4162
+ inserted: false,
4163
+ reason: "skipped-existing",
4164
+ provider: providerName,
4165
+ entries: existing.map(row => ({
4166
+ id: row.id,
4167
+ provider: providerName,
4168
+ credential: row.credential,
4169
+ disabledCause: null,
4170
+ })),
4171
+ };
4109
4172
  }
4110
4173
 
4111
4174
  this.#insertStmt.get(providerName, record.credentialType, record.data, record.identityKey);
@@ -4118,7 +4181,9 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
4118
4181
  },
4119
4182
  );
4120
4183
 
4121
- return writeIfAbsent.immediate(provider, serialized);
4184
+ const result = writeIfAbsent.immediate(provider, credential, serialized);
4185
+ if (result.inserted) this.#purgeSupersededDisabledRows(provider, result.entries);
4186
+ return result;
4122
4187
  }
4123
4188
 
4124
4189
  /**
@@ -4215,6 +4280,13 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
4215
4280
  }
4216
4281
  }
4217
4282
 
4283
+ deleteCachePrefix(prefix: string): void {
4284
+ if (prefix.length === 0) return;
4285
+ try {
4286
+ this.#deleteCachePrefixStmt.run(prefix.length, prefix);
4287
+ } catch {}
4288
+ }
4289
+
4218
4290
  cleanExpiredCache(): void {
4219
4291
  try {
4220
4292
  this.#deleteExpiredCacheStmt.run();
@@ -4305,6 +4377,7 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
4305
4377
  this.#getCacheStmt.finalize();
4306
4378
  this.#getCacheIncludingExpiredStmt.finalize();
4307
4379
  this.#upsertCacheStmt.finalize();
4380
+ this.#deleteCachePrefixStmt.finalize();
4308
4381
  this.#deleteExpiredCacheStmt.finalize();
4309
4382
  this.#db.close();
4310
4383
  }
@@ -13,6 +13,7 @@ import {
13
13
  anthropicModelManagerOptions,
14
14
  cerebrasModelManagerOptions,
15
15
  cloudflareAiGatewayModelManagerOptions,
16
+ deepinfraModelManagerOptions,
16
17
  deepseekModelManagerOptions,
17
18
  firepassModelManagerOptions,
18
19
  fireworksModelManagerOptions,
@@ -168,6 +169,12 @@ export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [
168
169
  config => deepseekModelManagerOptions(config),
169
170
  catalog("DeepSeek", ["DEEPSEEK_API_KEY"]),
170
171
  ),
172
+ catalogDescriptor(
173
+ "deepinfra",
174
+ "deepseek-ai/DeepSeek-V3.2",
175
+ config => deepinfraModelManagerOptions(config),
176
+ catalog("DeepInfra", ["DEEPINFRA_API_KEY"]),
177
+ ),
171
178
  descriptor("mistral", "devstral-medium-latest", config => mistralModelManagerOptions(config)),
172
179
  catalogDescriptor(
173
180
  "nvidia",
@@ -665,6 +665,21 @@ export function deepseekModelManagerOptions(
665
665
  ): ModelManagerOptions<"openai-completions"> {
666
666
  return createSimpleOpenAICompletionsOptions("deepseek", "https://api.deepseek.com", config);
667
667
  }
668
+ export interface DeepInfraModelManagerConfig {
669
+ apiKey?: string;
670
+ baseUrl?: string;
671
+ }
672
+
673
+ export function deepinfraModelManagerOptions(
674
+ config?: DeepInfraModelManagerConfig,
675
+ ): ModelManagerOptions<"openai-completions"> {
676
+ return createSimpleOpenAICompletionsOptions(
677
+ "deepinfra" as Parameters<typeof getBundledModels>[0],
678
+ "https://api.deepinfra.com/v1/openai",
679
+ config,
680
+ );
681
+ }
682
+
668
683
  // ---------------------------------------------------------------------------
669
684
  // 7.5 Fireworks
670
685
  // ---------------------------------------------------------------------------
@@ -2254,6 +2269,8 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
2254
2269
  }),
2255
2270
  // --- xAI ---
2256
2271
  openAiCompletionsDescriptor("xai", "xai", "https://api.x.ai/v1"),
2272
+ // --- DeepInfra ---
2273
+ openAiCompletionsDescriptor("deepinfra", "deepinfra", "https://api.deepinfra.com/v1/openai"),
2257
2274
  // --- DeepSeek ---
2258
2275
  openAiCompletionsDescriptor("deepseek", "deepseek", "https://api.deepseek.com", {
2259
2276
  // Only ship the v4 family as built-ins; older deepseek-chat / deepseek-reasoner
@@ -2281,8 +2281,6 @@ function buildParams(
2281
2281
  params.output_config = { effort } as typeof params.output_config;
2282
2282
  }
2283
2283
  }
2284
- } else if (options?.thinkingEnabled === false) {
2285
- params.thinking = { type: "disabled" };
2286
2284
  }
2287
2285
  }
2288
2286
 
@@ -35,6 +35,7 @@ import {
35
35
  markToolChoiceIncapability,
36
36
  resolveToolChoice,
37
37
  } from "../utils/tool-choice-capability";
38
+ import { wrapOpenAIFetchForBoundedRateLimits } from "./openai-bounded-rate-limits";
38
39
  import { normalizeOpenAIResponsesPromptCacheKey, supportsDeveloperRole } from "./openai-responses";
39
40
  import {
40
41
  appendResponsesToolResultMessages,
@@ -272,7 +273,7 @@ function createClient(model: Model<"azure-openai-responses">, apiKey: string, op
272
273
 
273
274
  const { baseUrl, apiVersion } = resolveAzureConfig(model, options);
274
275
 
275
- const baseFetch = options?.fetch ?? fetch;
276
+ const baseFetch = wrapOpenAIFetchForBoundedRateLimits(options?.fetch ?? fetch, options?.maxRetryDelayMs);
276
277
  const onSseEvent = options?.onSseEvent;
277
278
  return new AzureOpenAI({
278
279
  apiKey,
@@ -0,0 +1,10 @@
1
+ /**
2
+ * The Cursor client version reported to api2.cursor.sh.
3
+ *
4
+ * Every call against the Cursor backend must send the same
5
+ * x-cursor-client-version: the backend gates features and minimum versions on
6
+ * it, so a drift between the agent Run path and model discovery makes one of
7
+ * them fail while the other keeps working. Keep this as the single source of
8
+ * truth for the header value.
9
+ */
10
+ export const CURSOR_CLIENT_VERSION = "cli-2026.02.13-41ac335";
@@ -31,6 +31,7 @@ import { parseStreamingJson } from "../utils/json-parse";
31
31
  import { formatErrorMessageWithRetryAfter } from "../utils/retry-after";
32
32
  import { flattenToolRootCombinators, toolWireSchema } from "../utils/schema";
33
33
  import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
34
+ import { CURSOR_CLIENT_VERSION } from "./cursor/client-version";
34
35
  import type { McpToolDefinition } from "./cursor/gen/agent_pb";
35
36
  import {
36
37
  AgentClientMessageSchema,
@@ -131,7 +132,7 @@ import {
131
132
  } from "./cursor/gen/agent_pb";
132
133
 
133
134
  export const CURSOR_API_URL = "https://api2.cursor.sh";
134
- export const CURSOR_CLIENT_VERSION = "cli-2026.01.09-231024f";
135
+ export { CURSOR_CLIENT_VERSION };
135
136
 
136
137
  const conversationStateCache = new Map<string, ConversationStateStructure>();
137
138
  const conversationBlobStores = new Map<string, Map<string, Uint8Array>>();
@@ -4,7 +4,7 @@
4
4
  * GeminiCLI/VERSION/MODEL (PLATFORM; ARCH; SURFACE)
5
5
  */
6
6
  export function getGeminiCliUserAgent(modelId = "gemini-3.1-pro-preview"): string {
7
- const version = process.env.PI_AI_GEMINI_CLI_VERSION || "0.46.0";
7
+ const version = process.env.SKC_AI_GEMINI_CLI_VERSION || process.env.PI_AI_GEMINI_CLI_VERSION || "0.49.0";
8
8
  const platform = process.platform === "win32" ? "win32" : process.platform;
9
9
  const arch = process.arch === "x64" ? "x64" : process.arch;
10
10
  return `GeminiCLI/${version}/${modelId} (${platform}; ${arch}; terminal)`;
@@ -0,0 +1,57 @@
1
+ import type { FetchImpl } from "../types";
2
+ import { getRetryAfterMsFromHeaders } from "../utils/retry-after";
3
+
4
+ const OPENAI_RETRY_DELAY_CAP_MS = 60_000;
5
+
6
+ // Mirror of `wrapAnthropicFetchForBoundedRateLimits`: OpenAI-compatible providers
7
+ // (e.g. opencode-go) return HTTP 429 for *permanent* usage/quota exhaustion — a
8
+ // monthly-cap reset that can be days away. The OpenAI SDK treats 429 as transient
9
+ // and retries up to `maxRetries`, honoring an out-of-range `Retry-After`; the
10
+ // `create()` call then hangs before the error can surface to the agent loop, so
11
+ // no assistant error is produced and the session-level retry/fallback never runs.
12
+ // Detect exhaustion and set `x-should-retry: false` so the SDK gives up at once
13
+ // and the session retry layer applies its own fail-fast (retry-after > maxDelayMs).
14
+ //
15
+ // Shared by every adapter that drives a raw OpenAI SDK client — openai-completions,
16
+ // openai-responses, and azure-openai-responses. Adapters that route through
17
+ // `fetchWithRetry` (codex, bedrock, ollama, gemini-cli) already bound 429 retries
18
+ // themselves and do not need this wrapper.
19
+ export function isOpenAIUsageExhaustionResponse(
20
+ bodyText: string,
21
+ retryAfterMs: number | undefined,
22
+ retryDelayCapMs: number,
23
+ ): boolean {
24
+ if (retryAfterMs !== undefined && retryAfterMs > retryDelayCapMs) return true;
25
+ return /monthly usage limit|usage limit reached|usage_limit_reached|out_of_credits|insufficient_quota|quota[ _]?exceeded/i.test(
26
+ bodyText,
27
+ );
28
+ }
29
+
30
+ export function wrapOpenAIFetchForBoundedRateLimits(
31
+ baseFetch: FetchImpl,
32
+ maxRetryDelayMs: number | undefined,
33
+ ): FetchImpl {
34
+ const retryDelayCapMs = maxRetryDelayMs ?? OPENAI_RETRY_DELAY_CAP_MS;
35
+ return Object.assign(
36
+ async (input: string | URL | Request, init?: RequestInit): Promise<Response> => {
37
+ const response = await baseFetch(input, init);
38
+ if (response.status !== 429 || retryDelayCapMs === 0) return response;
39
+
40
+ const headers = new Headers(response.headers);
41
+ const retryAfterMs = getRetryAfterMsFromHeaders(headers);
42
+ const bodyText = await response
43
+ .clone()
44
+ .text()
45
+ .catch(() => "");
46
+ if (!isOpenAIUsageExhaustionResponse(bodyText, retryAfterMs, retryDelayCapMs)) return response;
47
+
48
+ headers.set("x-should-retry", "false");
49
+ return new Response(bodyText, {
50
+ status: response.status,
51
+ statusText: response.statusText,
52
+ headers,
53
+ });
54
+ },
55
+ baseFetch.preconnect ? { preconnect: baseFetch.preconnect } : {},
56
+ );
57
+ }
@@ -104,6 +104,10 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
104
104
  baseUrl.includes("opencode.ai");
105
105
  const isOpenCodeProvider = provider === "opencode-go" || provider === "opencode-zen";
106
106
  const isOpenCodeGoReasoning = provider === "opencode-go" && Boolean(model.reasoning);
107
+ const isOpenCodeGoKimiReasoning = provider === "opencode-go" && isKimiModel && Boolean(model.reasoning);
108
+ const isOpenCodeGoKimi25Reasoning = isOpenCodeGoKimiReasoning && model.id === "kimi-k2.5";
109
+ const isOpenCodeGoKimi27CodeReasoning = isOpenCodeGoKimiReasoning && model.id === "kimi-k2.7-code";
110
+ const needsOpenCodeGoKimiEffortMap = isOpenCodeGoKimi25Reasoning || isOpenCodeGoKimi27CodeReasoning;
107
111
 
108
112
  const useMaxTokens =
109
113
  provider === "mistral" ||
@@ -170,22 +174,31 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
170
174
  xhigh: "default",
171
175
  max: "default",
172
176
  } satisfies Partial<Record<OpenAIReasoningEffort, string>>)
173
- : isDeepseekFamily && model.reasoning
177
+ : needsOpenCodeGoKimiEffortMap
174
178
  ? ({
175
- minimal: "high",
176
- low: "high",
177
- medium: "high",
178
- high: "high",
179
- xhigh: "max",
180
- max: "max",
179
+ // Live Go probes (2026-07-06) showed model-specific effort gaps:
180
+ // kimi-k2.5 rejects "minimal", while kimi-k2.7-code rejects
181
+ // OpenAI-style "xhigh" and "max"; all other Kimi efforts tested
182
+ // successfully and should pass through unchanged.
183
+ ...(isOpenCodeGoKimi25Reasoning ? { minimal: "low" } : {}),
184
+ ...(isOpenCodeGoKimi27CodeReasoning ? { xhigh: "high", max: "high" } : {}),
181
185
  } satisfies Partial<Record<OpenAIReasoningEffort, string>>)
182
- : isFireworks
186
+ : isDeepseekFamily && model.reasoning
183
187
  ? ({
184
- // Fireworks' OpenAI-compatible endpoint rejects OpenAI's
185
- // `minimal` literal but accepts `none` for the lowest setting.
186
- minimal: "none",
188
+ minimal: "high",
189
+ low: "high",
190
+ medium: "high",
191
+ high: "high",
192
+ xhigh: "max",
193
+ max: "max",
187
194
  } satisfies Partial<Record<OpenAIReasoningEffort, string>>)
188
- : {};
195
+ : isFireworks
196
+ ? ({
197
+ // Fireworks' OpenAI-compatible endpoint rejects OpenAI's
198
+ // `minimal` literal but accepts `none` for the lowest setting.
199
+ minimal: "none",
200
+ } satisfies Partial<Record<OpenAIReasoningEffort, string>>)
201
+ : {};
189
202
 
190
203
  return {
191
204
  supportsStore: !isNonStandard,
@@ -198,7 +211,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
198
211
  disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel || isOpenCodeGoReasoning,
199
212
  disableReasoningOnToolChoice: isDeepseekFamily && Boolean(model.reasoning) && !isOpenRouter,
200
213
  supportsToolChoice: !isDirectDeepseekReasoning,
201
- supportsForcedToolChoice: true,
214
+ supportsForcedToolChoice: !isOpenCodeGoKimiReasoning,
202
215
  maxTokensField: useMaxTokens ? "max_tokens" : "max_completion_tokens",
203
216
  requiresToolResultName: isMistral,
204
217
  requiresAssistantAfterToolResult: false,
@@ -35,7 +35,7 @@ import {
35
35
  type ToolChoice,
36
36
  type ToolResultMessage,
37
37
  } from "../types";
38
- import { normalizeSystemPrompts } from "../utils";
38
+ import { normalizeSystemPrompts, sanitizeJsonStrings } from "../utils";
39
39
  import { createAbortSourceTracker } from "../utils/abort";
40
40
  import { AssistantMessageEventStream } from "../utils/event-stream";
41
41
  import { toFirepassWireModelId, toFireworksWireModelId } from "../utils/fireworks-model-id";
@@ -166,7 +166,7 @@ function normalizeStreamingContentText(content: unknown): string {
166
166
  function serializeToolArguments(value: unknown): string {
167
167
  if (value && typeof value === "object" && !Array.isArray(value)) {
168
168
  try {
169
- return JSON.stringify(value);
169
+ return JSON.stringify(sanitizeJsonStrings(value));
170
170
  } catch {
171
171
  return "{}";
172
172
  }
@@ -178,7 +178,7 @@ function serializeToolArguments(value: unknown): string {
178
178
  try {
179
179
  const parsed = JSON.parse(trimmed);
180
180
  if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) {
181
- return JSON.stringify(parsed);
181
+ return JSON.stringify(sanitizeJsonStrings(parsed));
182
182
  }
183
183
  } catch {}
184
184
  return "{}";
@@ -71,6 +71,7 @@ import {
71
71
  resolveGitHubCopilotBaseUrl,
72
72
  } from "./github-copilot-headers";
73
73
  import { compactGrammarDefinition } from "./grammar";
74
+ import { wrapOpenAIFetchForBoundedRateLimits } from "./openai-bounded-rate-limits";
74
75
  import {
75
76
  applyOpenAIRequestTransformBody,
76
77
  applyOpenAIRequestTransformHeaders,
@@ -274,6 +275,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
274
275
  options?.fetch,
275
276
  options?.authCredentialType,
276
277
  options?.requestMaxRetries,
278
+ options?.maxRetryDelayMs,
277
279
  );
278
280
  const premiumRequestsTotal = copilotPremiumRequests;
279
281
  const providerSessionState = getOpenAIResponsesProviderSessionState(model, options?.providerSessionState);
@@ -400,6 +402,7 @@ function createClient(
400
402
  fetchOverride?: FetchImpl,
401
403
  authCredentialType?: OpenAIResponsesOptions["authCredentialType"],
402
404
  requestMaxRetries?: number,
405
+ maxRetryDelayMs?: number,
403
406
  ): {
404
407
  client: OpenAI;
405
408
  copilotPremiumRequests: number | undefined;
@@ -446,8 +449,9 @@ function createClient(
446
449
  headers["x-client-request-id"] ??= sessionId;
447
450
  }
448
451
  const baseFetch = fetchOverride ?? fetch;
452
+ const boundedFetch = wrapOpenAIFetchForBoundedRateLimits(baseFetch, maxRetryDelayMs);
449
453
  const transformedFetch = wrapFetchForOpenAIRequestTransform(
450
- baseFetch,
454
+ boundedFetch,
451
455
  model.requestTransform,
452
456
  `Sayknow-CLI/${packageJson.version}`,
453
457
  );
@@ -13,7 +13,7 @@ export type RateLimitReason =
13
13
  const QUOTA_EXHAUSTED_BACKOFF_MS = 30 * 60 * 1000; // 30 min
14
14
  const RATE_LIMIT_EXCEEDED_BACKOFF_MS = 30 * 1000; // 30s
15
15
  const MODEL_CAPACITY_BASE_MS = 45 * 1000; // 45s base
16
- const MODEL_CAPACITY_JITTER_MS = 30 * 1000; // ±15s
16
+ const MODEL_CAPACITY_JITTER_MS = 30 * 1000; // uniform +0–30s above base → 45–75s total
17
17
  const SERVER_ERROR_BACKOFF_MS = 20 * 1000; // 20s
18
18
 
19
19
  /**
@@ -93,8 +93,10 @@ export function calculateRateLimitBackoffMs(reason: RateLimitReason): number {
93
93
  }
94
94
 
95
95
  /** Detect usage/quota limit errors in error messages (persistent, requires credential switch). */
96
+ // ZAI reports durable token exhaustion as "[1310][Weekly/Monthly Limit Exhausted...]".
97
+ // Keep this explicit so generic "rate limit exhausted, retry..." throttles remain retryable.
96
98
  const USAGE_LIMIT_PATTERN =
97
- /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|model.?limit|model_limit_reached|message.?limit|message_limit_reached|limit for this model|quota.?exceeded|out_of_credits|request would exceed your account.?s rate limit|resource has been exhausted[^\n]*(?:quota|limit)/i;
99
+ /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|model.?limit|model_limit_reached|message.?limit|message_limit_reached|limit for this model|weekly\/monthly\s+limit\s+exhausted|quota.?exceeded|out_of_credits|request would exceed your account.?s rate limit|resource has been exhausted[^\n]*(?:quota|limit)/i;
98
100
  export function isUsageLimitError(errorMessage: string): boolean {
99
101
  return USAGE_LIMIT_PATTERN.test(errorMessage);
100
102
  }
package/src/stream.ts CHANGED
@@ -98,6 +98,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
98
98
  "opencode-zen": "OPENCODE_API_KEY",
99
99
  cursor: "CURSOR_ACCESS_TOKEN",
100
100
  deepseek: "DEEPSEEK_API_KEY",
101
+ deepinfra: "DEEPINFRA_API_KEY",
101
102
  "openai-codex": "OPENAI_CODEX_OAUTH_TOKEN",
102
103
  "azure-openai": "AZURE_OPENAI_API_KEY",
103
104
  "azure-openai-responses": "AZURE_OPENAI_API_KEY",
package/src/types.ts CHANGED
@@ -116,6 +116,7 @@ export type KnownProvider =
116
116
  | "gitlab-duo"
117
117
  | "cursor"
118
118
  | "deepseek"
119
+ | "deepinfra"
119
120
  | "xai"
120
121
  | "groq"
121
122
  | "cerebras"
@@ -224,17 +225,20 @@ export function resolveServiceTier(
224
225
 
225
226
  /**
226
227
  * True when the (possibly scoped) tier should be sent as OpenAI's
227
- * `service_tier` request field for the given provider. Non-OpenAI
228
- * providers, unsupported tiers (`"auto"`, `"default"`), and scope
229
- * mismatches all return false.
228
+ * `service_tier` request field for the given provider. OpenAI accepts
229
+ * `flex`, `scale`, and `priority`; DeepInfra accepts `priority`.
230
+ * Unsupported tiers (`"auto"`, `"default"`) and scope mismatches return false.
230
231
  */
231
232
  export function shouldSendServiceTier(
232
233
  serviceTier: ServiceTier | null | undefined,
233
234
  provider: Provider | undefined,
234
235
  ): boolean {
235
- if (provider !== "openai" && provider !== "openai-codex") return false;
236
236
  const resolved = resolveServiceTier(serviceTier, provider);
237
- return resolved === "flex" || resolved === "scale" || resolved === "priority";
237
+ if (provider === "openai" || provider === "openai-codex") {
238
+ return resolved === "flex" || resolved === "scale" || resolved === "priority";
239
+ }
240
+ if (provider === "deepinfra") return resolved === "priority";
241
+ return false;
238
242
  }
239
243
 
240
244
  /**
@@ -252,7 +256,9 @@ export function getPriorityPremiumRequests(
252
256
  if (resolveServiceTier(serviceTier, provider) !== "priority") return 0;
253
257
  // Only providers that realize `priority` on the wire bill the user.
254
258
  // Everywhere else, the field is silently dropped and nothing is charged.
255
- return provider === "openai" || provider === "openai-codex" || provider === "anthropic" ? 1 : 0;
259
+ return provider === "openai" || provider === "openai-codex" || provider === "anthropic" || provider === "deepinfra"
260
+ ? 1
261
+ : 0;
256
262
  }
257
263
 
258
264
  export interface ProviderSessionState {
@@ -2,11 +2,11 @@ import * as http2 from "node:http2";
2
2
  import { create, fromBinary, toBinary } from "@bufbuild/protobuf";
3
3
  import * as z from "zod/v4";
4
4
  import { getBundledModels } from "../../models";
5
+ import { CURSOR_CLIENT_VERSION } from "../../providers/cursor/client-version";
5
6
  import { GetUsableModelsRequestSchema, GetUsableModelsResponseSchema } from "../../providers/cursor/gen/agent_pb";
6
7
  import type { Model } from "../../types";
7
8
 
8
9
  const CURSOR_DEFAULT_BASE_URL = "https://api2.cursor.sh";
9
- const CURSOR_DEFAULT_CLIENT_VERSION = "cli-2026.02.13-41ac335";
10
10
  const CURSOR_GET_USABLE_MODELS_PATH = "/agent.v1.AgentService/GetUsableModels";
11
11
 
12
12
  const DEFAULT_CONTEXT_WINDOW = 200_000;
@@ -91,7 +91,7 @@ function buildRequestHeaders(options: CursorModelDiscoveryOptions): Record<strin
91
91
  te: "trailers",
92
92
  authorization: `Bearer ${options.apiKey}`,
93
93
  "x-ghost-mode": "true",
94
- "x-cursor-client-version": options.clientVersion ?? CURSOR_DEFAULT_CLIENT_VERSION,
94
+ "x-cursor-client-version": options.clientVersion ?? CURSOR_CLIENT_VERSION,
95
95
  "x-cursor-client-type": "cli",
96
96
  };
97
97
  }
@@ -3,10 +3,10 @@ import { createApiKeyLogin } from "./api-key-login";
3
3
 
4
4
  export const loginFugu = createApiKeyLogin({
5
5
  providerLabel: "Sakana Fugu",
6
- authUrl: "https://fugu.sakana.ai/",
6
+ authUrl: "https://console.sakana.ai/api-keys",
7
7
  instructions: "Create or copy your Sakana Fugu API key",
8
8
  promptMessage: "Paste your Sakana Fugu API key",
9
- placeholder: "fugu_...",
9
+ placeholder: "fish_...",
10
10
  validation: {
11
11
  kind: "models-endpoint",
12
12
  provider: "Sakana Fugu",
@@ -60,6 +60,11 @@ const builtInOAuthProviders: OAuthProviderInfo[] = [
60
60
  name: "DeepSeek",
61
61
  available: true,
62
62
  },
63
+ {
64
+ id: "deepinfra",
65
+ name: "DeepInfra",
66
+ available: true,
67
+ },
63
68
  {
64
69
  id: "xai",
65
70
  name: "xAI",
@@ -359,6 +364,7 @@ export async function refreshOAuthToken(
359
364
  case "fireworks":
360
365
  case "firepass":
361
366
  case "fugu":
367
+ case "deepinfra":
362
368
  case "nvidia":
363
369
  case "nanogpt":
364
370
  case "synthetic":
@@ -15,6 +15,7 @@ export type OAuthProvider =
15
15
  | "cloudflare-ai-gateway"
16
16
  | "cursor"
17
17
  | "deepseek"
18
+ | "deepinfra"
18
19
  | "fireworks"
19
20
  | "firepass"
20
21
  | "fugu"
package/src/utils.ts CHANGED
@@ -215,7 +215,11 @@ function sanitizeOpenAIResponsesHistoryItemForReplay(
215
215
  // providerPayload stores raw output items; replay strips fields that are output-only.
216
216
  const { id: _id, ...itemWithoutId } = item;
217
217
  const sanitizedItem =
218
- item.type === "computer_call" ? sanitizeComputerCallForResponsesInput(itemWithoutId) : itemWithoutId;
218
+ item.type === "computer_call"
219
+ ? sanitizeComputerCallForResponsesInput(itemWithoutId)
220
+ : item.type === "image_generation_call"
221
+ ? sanitizeImageGenerationCallForResponsesInput(itemWithoutId)
222
+ : itemWithoutId;
219
223
  if (typeof item.call_id === "string") {
220
224
  sanitizedItem.call_id = normalizeReplayedResponsesHistoryCallId(item.call_id, normalizedCallIds);
221
225
  }
@@ -231,6 +235,22 @@ function sanitizeComputerCallForResponsesInput(item: Record<string, unknown>): R
231
235
  return inputSafeItem;
232
236
  }
233
237
 
238
+ function sanitizeImageGenerationCallForResponsesInput(item: Record<string, unknown>): Record<string, unknown> {
239
+ // Image generation output items include request-time knobs that are not part of
240
+ // the Responses input replay schema. Replaying them verbatim makes OpenAI-compatible
241
+ // endpoints reject the next turn, e.g. `Unknown parameter: input[n].action`.
242
+ const {
243
+ action: _action,
244
+ background: _background,
245
+ output_format: _outputFormat,
246
+ quality: _quality,
247
+ revised_prompt: _revisedPrompt,
248
+ size: _size,
249
+ ...inputSafeItem
250
+ } = item;
251
+ return inputSafeItem;
252
+ }
253
+
234
254
  function normalizeReplayedResponsesHistoryCallId(value: string, normalizedValues: Map<string, string>): string {
235
255
  const normalized = normalizedValues.get(value);
236
256
  if (normalized) return normalized;