@gajae-code/ai 0.7.8 → 0.7.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,12 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.7.9] - 2026-07-01
6
+
7
+ ### Fixed
8
+
9
+ - Mapped DeepSeek-style `prompt_cache_hit_tokens` and `prompt_cache_miss_tokens` usage fields into OpenAI-compatible prompt-cache accounting (#1329).
10
+
5
11
  ## [0.7.5] - 2026-06-27
6
12
 
7
13
  ### Fixed
@@ -58,6 +58,11 @@ export interface DeepSeekModelManagerConfig {
58
58
  baseUrl?: string;
59
59
  }
60
60
  export declare function deepseekModelManagerOptions(config?: DeepSeekModelManagerConfig): ModelManagerOptions<"openai-completions">;
61
+ export interface DeepInfraModelManagerConfig {
62
+ apiKey?: string;
63
+ baseUrl?: string;
64
+ }
65
+ export declare function deepinfraModelManagerOptions(config?: DeepInfraModelManagerConfig): ModelManagerOptions<"openai-completions">;
61
66
  export interface FireworksModelManagerConfig {
62
67
  apiKey?: string;
63
68
  baseUrl?: string;
@@ -134,6 +134,7 @@ export type AnthropicClientOptionsArgs = {
134
134
  onSseEvent?: AnthropicOptions["onSseEvent"];
135
135
  fetch?: FetchImpl;
136
136
  requestMaxRetries?: number;
137
+ maxRetryDelayMs?: number;
137
138
  };
138
139
  export type AnthropicClientOptionsResult = {
139
140
  isOAuthToken: boolean;
@@ -3,6 +3,9 @@
3
3
  * Uses the same format as the official Gemini CLI (v0.35+):
4
4
  * GeminiCLI/VERSION/MODEL (PLATFORM; ARCH; SURFACE)
5
5
  */
6
+ export declare const GEMINI_CLI_VERSION_ENV = "GJC_AI_GEMINI_CLI_VERSION";
7
+ export declare const LEGACY_GEMINI_CLI_VERSION_ENV = "PI_AI_GEMINI_CLI_VERSION";
8
+ export declare const DEFAULT_GEMINI_CLI_VERSION = "0.49.0";
6
9
  export declare function getGeminiCliUserAgent(modelId?: string): string;
7
10
  export declare const getGeminiCliHeaders: (modelId?: string) => {
8
11
  "User-Agent": string;
@@ -48,7 +48,7 @@ export interface ThinkingConfig {
48
48
  /** Provider-specific transport used to encode the selected effort. */
49
49
  mode: ThinkingControlMode;
50
50
  }
51
- export type KnownProvider = "alibaba-coding-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
51
+ export type KnownProvider = "alibaba-coding-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
52
52
  export type Provider = KnownProvider | string;
53
53
  import type { Effort } from "./model-thinking";
54
54
  /** Token budgets for each thinking level (token-based providers only) */
@@ -84,7 +84,7 @@ export type CacheRetention = "none" | "short" | "long";
84
84
  *
85
85
  * The unscoped values (`"auto"`, `"default"`, `"flex"`, `"scale"`,
86
86
  * `"priority"`) are passed through to providers that understand them
87
- * (OpenAI's `service_tier` field directly; Anthropic translates
87
+ * (OpenAI and DeepInfra's `service_tier` field directly; Anthropic translates
88
88
  * `"priority"` into `speed: "fast"` on supported Opus models).
89
89
  *
90
90
  * The scoped values target a specific provider family and behave as the
@@ -105,10 +105,9 @@ export type ResolvedServiceTier = Exclude<ServiceTier, "openai-only" | "claude-o
105
105
  */
106
106
  export declare function resolveServiceTier(serviceTier: ServiceTier | null | undefined, provider: Provider | undefined): ResolvedServiceTier | undefined;
107
107
  /**
108
- * True when the (possibly scoped) tier should be sent as OpenAI's
109
- * `service_tier` request field for the given provider. Non-OpenAI
110
- * providers, unsupported tiers (`"auto"`, `"default"`), and scope
111
- * mismatches all return false.
108
+ * True when the (possibly scoped) tier should be sent as an OpenAI-compatible
109
+ * `service_tier` request field for providers that support it. Unsupported tiers
110
+ * (`"auto"`, `"default"`) and scope mismatches all return false.
112
111
  */
113
112
  export declare function shouldSendServiceTier(serviceTier: ServiceTier | null | undefined, provider: Provider | undefined): boolean;
114
113
  /**
@@ -0,0 +1 @@
1
+ export declare const loginDeepInfra: (options: import("./types").OAuthController) => Promise<string>;
@@ -7,7 +7,7 @@ export type OAuthCredentials = {
7
7
  email?: string;
8
8
  accountId?: string;
9
9
  };
10
- export type OAuthProvider = "alibaba-coding-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
10
+ export type OAuthProvider = "alibaba-coding-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
11
11
  export type OAuthProviderId = OAuthProvider | (string & {});
12
12
  export type OAuthPrompt = {
13
13
  message: string;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/ai",
4
- "version": "0.7.8",
4
+ "version": "0.7.10",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://gaebal-gajae.dev",
7
7
  "author": "Yeachan-Heo",
@@ -43,7 +43,7 @@
43
43
  "dependencies": {
44
44
  "@anthropic-ai/sdk": "^0.94.0",
45
45
  "@bufbuild/protobuf": "^2.12.0",
46
- "@gajae-code/utils": "0.7.8",
46
+ "@gajae-code/utils": "0.7.10",
47
47
  "openai": "^6.36.0",
48
48
  "partial-json": "^0.1.7",
49
49
  "zod": "4.4.3"
@@ -32,6 +32,7 @@ import { kimiUsageProvider } from "./usage/kimi";
32
32
  import { codexRankingStrategy, openaiCodexUsageProvider } from "./usage/openai-codex";
33
33
  import { zaiUsageProvider } from "./usage/zai";
34
34
  import { getOAuthApiKey, getOAuthProvider, refreshOAuthToken, resolveOAuthStorageProvider } from "./utils/oauth";
35
+ import { loginDeepInfra } from "./utils/oauth/deepinfra";
35
36
  import { loginDeepSeek } from "./utils/oauth/deepseek";
36
37
  import { loginOpenAICodexDevice } from "./utils/oauth/openai-codex";
37
38
  import type { OAuthController, OAuthCredentials, OAuthProvider, OAuthProviderId } from "./utils/oauth/types";
@@ -1593,6 +1594,11 @@ export class AuthStorage {
1593
1594
  await saveApiKeyCredential(apiKey);
1594
1595
  return;
1595
1596
  }
1597
+ case "deepinfra": {
1598
+ const apiKey = await loginDeepInfra(ctrl);
1599
+ await saveApiKeyCredential(apiKey);
1600
+ return;
1601
+ }
1596
1602
  case "xai": {
1597
1603
  const { loginXai } = await import("./utils/oauth/xai");
1598
1604
  credentials = await loginXai({
package/src/cli.ts CHANGED
@@ -111,6 +111,7 @@ Providers:
111
111
  zai Z.AI (GLM Coding Plan)
112
112
  glm-zcode GLM ZCode OAuth (unofficial, opt-in; at your own risk)
113
113
  deepseek DeepSeek
114
+ deepinfra DeepInfra
114
115
  xai xAI
115
116
  nanogpt NanoGPT
116
117
  minimax-code MiniMax Coding Plan (International)
package/src/models.json CHANGED
@@ -3583,6 +3583,31 @@
3583
3583
  "minLevel": "minimal",
3584
3584
  "maxLevel": "high"
3585
3585
  }
3586
+ },
3587
+ "claude-sonnet-5": {
3588
+ "id": "claude-sonnet-5",
3589
+ "name": "Anthropic Sonnet 5",
3590
+ "api": "anthropic-messages",
3591
+ "provider": "anthropic",
3592
+ "baseUrl": "https://api.anthropic.com",
3593
+ "reasoning": true,
3594
+ "input": [
3595
+ "text",
3596
+ "image"
3597
+ ],
3598
+ "cost": {
3599
+ "input": 3,
3600
+ "output": 15,
3601
+ "cacheRead": 0.3,
3602
+ "cacheWrite": 3.75
3603
+ },
3604
+ "contextWindow": 1000000,
3605
+ "maxTokens": 64000,
3606
+ "thinking": {
3607
+ "mode": "anthropic-adaptive",
3608
+ "minLevel": "minimal",
3609
+ "maxLevel": "high"
3610
+ }
3586
3611
  }
3587
3612
  },
3588
3613
  "azure-openai": {
@@ -7722,6 +7747,602 @@
7722
7747
  }
7723
7748
  }
7724
7749
  },
7750
+ "deepinfra": {
7751
+ "deepseek-ai/DeepSeek-R1-0528": {
7752
+ "id": "deepseek-ai/DeepSeek-R1-0528",
7753
+ "name": "DeepSeek-R1-0528",
7754
+ "api": "openai-completions",
7755
+ "provider": "deepinfra",
7756
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
7757
+ "reasoning": true,
7758
+ "input": [
7759
+ "text"
7760
+ ],
7761
+ "cost": {
7762
+ "input": 0.5,
7763
+ "output": 2.15,
7764
+ "cacheRead": 0.35,
7765
+ "cacheWrite": 0
7766
+ },
7767
+ "contextWindow": 163840,
7768
+ "maxTokens": 64000,
7769
+ "thinking": {
7770
+ "mode": "effort",
7771
+ "minLevel": "minimal",
7772
+ "maxLevel": "xhigh"
7773
+ }
7774
+ },
7775
+ "deepseek-ai/DeepSeek-V3.2": {
7776
+ "id": "deepseek-ai/DeepSeek-V3.2",
7777
+ "name": "DeepSeek-V3.2",
7778
+ "api": "openai-completions",
7779
+ "provider": "deepinfra",
7780
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
7781
+ "reasoning": true,
7782
+ "input": [
7783
+ "text"
7784
+ ],
7785
+ "cost": {
7786
+ "input": 0.26,
7787
+ "output": 0.38,
7788
+ "cacheRead": 0.13,
7789
+ "cacheWrite": 0
7790
+ },
7791
+ "contextWindow": 163840,
7792
+ "maxTokens": 64000,
7793
+ "thinking": {
7794
+ "mode": "effort",
7795
+ "minLevel": "minimal",
7796
+ "maxLevel": "xhigh"
7797
+ }
7798
+ },
7799
+ "deepseek-ai/DeepSeek-V4-Flash": {
7800
+ "id": "deepseek-ai/DeepSeek-V4-Flash",
7801
+ "name": "DeepSeek V4 Flash",
7802
+ "api": "openai-completions",
7803
+ "provider": "deepinfra",
7804
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
7805
+ "reasoning": true,
7806
+ "input": [
7807
+ "text"
7808
+ ],
7809
+ "cost": {
7810
+ "input": 0.1,
7811
+ "output": 0.2,
7812
+ "cacheRead": 0.02,
7813
+ "cacheWrite": 0
7814
+ },
7815
+ "contextWindow": 1048576,
7816
+ "maxTokens": 16384,
7817
+ "thinking": {
7818
+ "mode": "effort",
7819
+ "minLevel": "minimal",
7820
+ "maxLevel": "xhigh"
7821
+ }
7822
+ },
7823
+ "deepseek-ai/DeepSeek-V4-Pro": {
7824
+ "id": "deepseek-ai/DeepSeek-V4-Pro",
7825
+ "name": "DeepSeek V4 Pro",
7826
+ "api": "openai-completions",
7827
+ "provider": "deepinfra",
7828
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
7829
+ "reasoning": true,
7830
+ "input": [
7831
+ "text"
7832
+ ],
7833
+ "cost": {
7834
+ "input": 1.3,
7835
+ "output": 2.6,
7836
+ "cacheRead": 0.1,
7837
+ "cacheWrite": 0
7838
+ },
7839
+ "contextWindow": 1048576,
7840
+ "maxTokens": 16384,
7841
+ "thinking": {
7842
+ "mode": "effort",
7843
+ "minLevel": "minimal",
7844
+ "maxLevel": "xhigh"
7845
+ }
7846
+ },
7847
+ "google/gemma-4-26B-A4B-it": {
7848
+ "id": "google/gemma-4-26B-A4B-it",
7849
+ "name": "Gemma 4 26B A4B IT",
7850
+ "api": "openai-completions",
7851
+ "provider": "deepinfra",
7852
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
7853
+ "reasoning": true,
7854
+ "input": [
7855
+ "text",
7856
+ "image"
7857
+ ],
7858
+ "cost": {
7859
+ "input": 0.07,
7860
+ "output": 0.34,
7861
+ "cacheRead": 0,
7862
+ "cacheWrite": 0
7863
+ },
7864
+ "contextWindow": 262144,
7865
+ "maxTokens": 32768,
7866
+ "thinking": {
7867
+ "mode": "effort",
7868
+ "minLevel": "minimal",
7869
+ "maxLevel": "xhigh"
7870
+ }
7871
+ },
7872
+ "google/gemma-4-31B-it": {
7873
+ "id": "google/gemma-4-31B-it",
7874
+ "name": "Gemma 4 31B IT",
7875
+ "api": "openai-completions",
7876
+ "provider": "deepinfra",
7877
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
7878
+ "reasoning": true,
7879
+ "input": [
7880
+ "text",
7881
+ "image"
7882
+ ],
7883
+ "cost": {
7884
+ "input": 0.13,
7885
+ "output": 0.38,
7886
+ "cacheRead": 0,
7887
+ "cacheWrite": 0
7888
+ },
7889
+ "contextWindow": 262144,
7890
+ "maxTokens": 32768,
7891
+ "thinking": {
7892
+ "mode": "effort",
7893
+ "minLevel": "minimal",
7894
+ "maxLevel": "xhigh"
7895
+ }
7896
+ },
7897
+ "meta-llama/Llama-3.3-70B-Instruct-Turbo": {
7898
+ "id": "meta-llama/Llama-3.3-70B-Instruct-Turbo",
7899
+ "name": "Llama 3.3 70B Turbo",
7900
+ "api": "openai-completions",
7901
+ "provider": "deepinfra",
7902
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
7903
+ "reasoning": false,
7904
+ "input": [
7905
+ "text"
7906
+ ],
7907
+ "cost": {
7908
+ "input": 0.1,
7909
+ "output": 0.32,
7910
+ "cacheRead": 0,
7911
+ "cacheWrite": 0
7912
+ },
7913
+ "contextWindow": 131072,
7914
+ "maxTokens": 16384
7915
+ },
7916
+ "meta-llama/Llama-4-Scout-17B-16E-Instruct": {
7917
+ "id": "meta-llama/Llama-4-Scout-17B-16E-Instruct",
7918
+ "name": "Llama 4 Scout 17B",
7919
+ "api": "openai-completions",
7920
+ "provider": "deepinfra",
7921
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
7922
+ "reasoning": false,
7923
+ "input": [
7924
+ "text",
7925
+ "image"
7926
+ ],
7927
+ "cost": {
7928
+ "input": 0.1,
7929
+ "output": 0.3,
7930
+ "cacheRead": 0,
7931
+ "cacheWrite": 0
7932
+ },
7933
+ "contextWindow": 327680,
7934
+ "maxTokens": 16384
7935
+ },
7936
+ "MiniMaxAI/MiniMax-M2.5": {
7937
+ "id": "MiniMaxAI/MiniMax-M2.5",
7938
+ "name": "MiniMax M2.5",
7939
+ "api": "openai-completions",
7940
+ "provider": "deepinfra",
7941
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
7942
+ "reasoning": true,
7943
+ "input": [
7944
+ "text"
7945
+ ],
7946
+ "cost": {
7947
+ "input": 0.15,
7948
+ "output": 1.15,
7949
+ "cacheRead": 0.03,
7950
+ "cacheWrite": 0.375
7951
+ },
7952
+ "contextWindow": 196608,
7953
+ "maxTokens": 131072,
7954
+ "thinking": {
7955
+ "mode": "effort",
7956
+ "minLevel": "minimal",
7957
+ "maxLevel": "xhigh"
7958
+ }
7959
+ },
7960
+ "moonshotai/Kimi-K2.5": {
7961
+ "id": "moonshotai/Kimi-K2.5",
7962
+ "name": "Kimi K2.5",
7963
+ "api": "openai-completions",
7964
+ "provider": "deepinfra",
7965
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
7966
+ "reasoning": true,
7967
+ "input": [
7968
+ "text",
7969
+ "image"
7970
+ ],
7971
+ "cost": {
7972
+ "input": 0.45,
7973
+ "output": 2.25,
7974
+ "cacheRead": 0.07,
7975
+ "cacheWrite": 0
7976
+ },
7977
+ "contextWindow": 262144,
7978
+ "maxTokens": 32768,
7979
+ "thinking": {
7980
+ "mode": "effort",
7981
+ "minLevel": "minimal",
7982
+ "maxLevel": "xhigh"
7983
+ }
7984
+ },
7985
+ "moonshotai/Kimi-K2.6": {
7986
+ "id": "moonshotai/Kimi-K2.6",
7987
+ "name": "Kimi K2.6",
7988
+ "api": "openai-completions",
7989
+ "provider": "deepinfra",
7990
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
7991
+ "reasoning": true,
7992
+ "input": [
7993
+ "text",
7994
+ "image"
7995
+ ],
7996
+ "cost": {
7997
+ "input": 0.75,
7998
+ "output": 3.5,
7999
+ "cacheRead": 0.15,
8000
+ "cacheWrite": 0
8001
+ },
8002
+ "contextWindow": 262144,
8003
+ "maxTokens": 16384,
8004
+ "thinking": {
8005
+ "mode": "effort",
8006
+ "minLevel": "minimal",
8007
+ "maxLevel": "xhigh"
8008
+ }
8009
+ },
8010
+ "openai/gpt-oss-120b": {
8011
+ "id": "openai/gpt-oss-120b",
8012
+ "name": "GPT OSS 120B",
8013
+ "api": "openai-completions",
8014
+ "provider": "deepinfra",
8015
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
8016
+ "reasoning": true,
8017
+ "input": [
8018
+ "text"
8019
+ ],
8020
+ "cost": {
8021
+ "input": 0.039,
8022
+ "output": 0.19,
8023
+ "cacheRead": 0,
8024
+ "cacheWrite": 0
8025
+ },
8026
+ "contextWindow": 131072,
8027
+ "maxTokens": 16384,
8028
+ "thinking": {
8029
+ "mode": "effort",
8030
+ "minLevel": "minimal",
8031
+ "maxLevel": "xhigh"
8032
+ }
8033
+ },
8034
+ "openai/gpt-oss-20b": {
8035
+ "id": "openai/gpt-oss-20b",
8036
+ "name": "GPT OSS 20B",
8037
+ "api": "openai-completions",
8038
+ "provider": "deepinfra",
8039
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
8040
+ "reasoning": true,
8041
+ "input": [
8042
+ "text"
8043
+ ],
8044
+ "cost": {
8045
+ "input": 0.03,
8046
+ "output": 0.14,
8047
+ "cacheRead": 0,
8048
+ "cacheWrite": 0
8049
+ },
8050
+ "contextWindow": 131072,
8051
+ "maxTokens": 16384,
8052
+ "thinking": {
8053
+ "mode": "effort",
8054
+ "minLevel": "minimal",
8055
+ "maxLevel": "xhigh"
8056
+ }
8057
+ },
8058
+ "Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo": {
8059
+ "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo",
8060
+ "name": "Qwen3 Coder 480B A35B Instruct Turbo",
8061
+ "api": "openai-completions",
8062
+ "provider": "deepinfra",
8063
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
8064
+ "reasoning": false,
8065
+ "input": [
8066
+ "text"
8067
+ ],
8068
+ "cost": {
8069
+ "input": 0.3,
8070
+ "output": 1,
8071
+ "cacheRead": 0,
8072
+ "cacheWrite": 0
8073
+ },
8074
+ "contextWindow": 262144,
8075
+ "maxTokens": 66536
8076
+ },
8077
+ "Qwen/Qwen3.5-35B-A3B": {
8078
+ "id": "Qwen/Qwen3.5-35B-A3B",
8079
+ "name": "Qwen 3.5 35B A3B",
8080
+ "api": "openai-completions",
8081
+ "provider": "deepinfra",
8082
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
8083
+ "reasoning": true,
8084
+ "input": [
8085
+ "text",
8086
+ "image"
8087
+ ],
8088
+ "cost": {
8089
+ "input": 0.14,
8090
+ "output": 1,
8091
+ "cacheRead": 0.05,
8092
+ "cacheWrite": 0
8093
+ },
8094
+ "contextWindow": 262144,
8095
+ "maxTokens": 81920,
8096
+ "thinking": {
8097
+ "mode": "effort",
8098
+ "minLevel": "minimal",
8099
+ "maxLevel": "high"
8100
+ }
8101
+ },
8102
+ "Qwen/Qwen3.5-397B-A17B": {
8103
+ "id": "Qwen/Qwen3.5-397B-A17B",
8104
+ "name": "Qwen 3.5 397B A17B",
8105
+ "api": "openai-completions",
8106
+ "provider": "deepinfra",
8107
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
8108
+ "reasoning": true,
8109
+ "input": [
8110
+ "text",
8111
+ "image"
8112
+ ],
8113
+ "cost": {
8114
+ "input": 0.45,
8115
+ "output": 3,
8116
+ "cacheRead": 0.22,
8117
+ "cacheWrite": 0
8118
+ },
8119
+ "contextWindow": 262144,
8120
+ "maxTokens": 81920,
8121
+ "thinking": {
8122
+ "mode": "effort",
8123
+ "minLevel": "minimal",
8124
+ "maxLevel": "high"
8125
+ }
8126
+ },
8127
+ "Qwen/Qwen3.6-35B-A3B": {
8128
+ "id": "Qwen/Qwen3.6-35B-A3B",
8129
+ "name": "Qwen3.6 35B A3B",
8130
+ "api": "openai-completions",
8131
+ "provider": "deepinfra",
8132
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
8133
+ "reasoning": true,
8134
+ "input": [
8135
+ "text",
8136
+ "image"
8137
+ ],
8138
+ "cost": {
8139
+ "input": 0.15,
8140
+ "output": 0.95,
8141
+ "cacheRead": 0,
8142
+ "cacheWrite": 0
8143
+ },
8144
+ "contextWindow": 262144,
8145
+ "maxTokens": 81920,
8146
+ "thinking": {
8147
+ "mode": "effort",
8148
+ "minLevel": "minimal",
8149
+ "maxLevel": "high"
8150
+ }
8151
+ },
8152
+ "XiaomiMiMo/MiMo-V2.5": {
8153
+ "id": "XiaomiMiMo/MiMo-V2.5",
8154
+ "name": "MiMo-V2.5",
8155
+ "api": "openai-completions",
8156
+ "provider": "deepinfra",
8157
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
8158
+ "reasoning": true,
8159
+ "input": [
8160
+ "text",
8161
+ "image"
8162
+ ],
8163
+ "cost": {
8164
+ "input": 0.4,
8165
+ "output": 2,
8166
+ "cacheRead": 0.08,
8167
+ "cacheWrite": 0
8168
+ },
8169
+ "contextWindow": 262144,
8170
+ "maxTokens": 16384,
8171
+ "thinking": {
8172
+ "mode": "effort",
8173
+ "minLevel": "minimal",
8174
+ "maxLevel": "xhigh"
8175
+ }
8176
+ },
8177
+ "XiaomiMiMo/MiMo-V2.5-Pro": {
8178
+ "id": "XiaomiMiMo/MiMo-V2.5-Pro",
8179
+ "name": "MiMo-V2.5-Pro",
8180
+ "api": "openai-completions",
8181
+ "provider": "deepinfra",
8182
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
8183
+ "reasoning": true,
8184
+ "input": [
8185
+ "text"
8186
+ ],
8187
+ "cost": {
8188
+ "input": 1,
8189
+ "output": 3,
8190
+ "cacheRead": 0.2,
8191
+ "cacheWrite": 0
8192
+ },
8193
+ "contextWindow": 1048576,
8194
+ "maxTokens": 16384,
8195
+ "thinking": {
8196
+ "mode": "effort",
8197
+ "minLevel": "minimal",
8198
+ "maxLevel": "xhigh"
8199
+ }
8200
+ },
8201
+ "zai-org/GLM-4.6": {
8202
+ "id": "zai-org/GLM-4.6",
8203
+ "name": "GLM-4.6",
8204
+ "api": "openai-completions",
8205
+ "provider": "deepinfra",
8206
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
8207
+ "reasoning": true,
8208
+ "input": [
8209
+ "text"
8210
+ ],
8211
+ "cost": {
8212
+ "input": 0.43,
8213
+ "output": 1.74,
8214
+ "cacheRead": 0.08,
8215
+ "cacheWrite": 0
8216
+ },
8217
+ "contextWindow": 202752,
8218
+ "maxTokens": 131072,
8219
+ "thinking": {
8220
+ "mode": "effort",
8221
+ "minLevel": "minimal",
8222
+ "maxLevel": "xhigh"
8223
+ }
8224
+ },
8225
+ "zai-org/GLM-4.7": {
8226
+ "id": "zai-org/GLM-4.7",
8227
+ "name": "GLM-4.7",
8228
+ "api": "openai-completions",
8229
+ "provider": "deepinfra",
8230
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
8231
+ "reasoning": true,
8232
+ "input": [
8233
+ "text"
8234
+ ],
8235
+ "cost": {
8236
+ "input": 0.4,
8237
+ "output": 1.75,
8238
+ "cacheRead": 0.08,
8239
+ "cacheWrite": 0
8240
+ },
8241
+ "contextWindow": 202752,
8242
+ "maxTokens": 16384,
8243
+ "thinking": {
8244
+ "mode": "effort",
8245
+ "minLevel": "minimal",
8246
+ "maxLevel": "xhigh"
8247
+ }
8248
+ },
8249
+ "zai-org/GLM-4.7-Flash": {
8250
+ "id": "zai-org/GLM-4.7-Flash",
8251
+ "name": "GLM-4.7-Flash",
8252
+ "api": "openai-completions",
8253
+ "provider": "deepinfra",
8254
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
8255
+ "reasoning": true,
8256
+ "input": [
8257
+ "text"
8258
+ ],
8259
+ "cost": {
8260
+ "input": 0.06,
8261
+ "output": 0.4,
8262
+ "cacheRead": 0,
8263
+ "cacheWrite": 0
8264
+ },
8265
+ "contextWindow": 202752,
8266
+ "maxTokens": 16384,
8267
+ "thinking": {
8268
+ "mode": "effort",
8269
+ "minLevel": "minimal",
8270
+ "maxLevel": "xhigh"
8271
+ }
8272
+ },
8273
+ "zai-org/GLM-5": {
8274
+ "id": "zai-org/GLM-5",
8275
+ "name": "GLM-5",
8276
+ "api": "openai-completions",
8277
+ "provider": "deepinfra",
8278
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
8279
+ "reasoning": true,
8280
+ "input": [
8281
+ "text"
8282
+ ],
8283
+ "cost": {
8284
+ "input": 0.6,
8285
+ "output": 2.08,
8286
+ "cacheRead": 0.12,
8287
+ "cacheWrite": 0
8288
+ },
8289
+ "contextWindow": 202752,
8290
+ "maxTokens": 16384,
8291
+ "thinking": {
8292
+ "mode": "effort",
8293
+ "minLevel": "minimal",
8294
+ "maxLevel": "xhigh"
8295
+ }
8296
+ },
8297
+ "zai-org/GLM-5.1": {
8298
+ "id": "zai-org/GLM-5.1",
8299
+ "name": "GLM-5.1",
8300
+ "api": "openai-completions",
8301
+ "provider": "deepinfra",
8302
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
8303
+ "reasoning": true,
8304
+ "input": [
8305
+ "text"
8306
+ ],
8307
+ "cost": {
8308
+ "input": 1.05,
8309
+ "output": 3.5,
8310
+ "cacheRead": 0.205,
8311
+ "cacheWrite": 0
8312
+ },
8313
+ "contextWindow": 202752,
8314
+ "maxTokens": 16384,
8315
+ "thinking": {
8316
+ "mode": "effort",
8317
+ "minLevel": "minimal",
8318
+ "maxLevel": "xhigh"
8319
+ }
8320
+ },
8321
+ "zai-org/GLM-5.2": {
8322
+ "id": "zai-org/GLM-5.2",
8323
+ "name": "GLM-5.2",
8324
+ "api": "openai-completions",
8325
+ "provider": "deepinfra",
8326
+ "baseUrl": "https://api.deepinfra.com/v1/openai",
8327
+ "reasoning": true,
8328
+ "input": [
8329
+ "text"
8330
+ ],
8331
+ "cost": {
8332
+ "input": 0.95,
8333
+ "output": 3,
8334
+ "cacheRead": 0.18,
8335
+ "cacheWrite": 0
8336
+ },
8337
+ "contextWindow": 1048576,
8338
+ "maxTokens": 32768,
8339
+ "thinking": {
8340
+ "mode": "effort",
8341
+ "minLevel": "minimal",
8342
+ "maxLevel": "xhigh"
8343
+ }
8344
+ }
8345
+ },
7725
8346
  "firepass": {
7726
8347
  "kimi-k2.6-turbo": {
7727
8348
  "id": "kimi-k2.6-turbo",
@@ -13,6 +13,7 @@ import {
13
13
  anthropicModelManagerOptions,
14
14
  cerebrasModelManagerOptions,
15
15
  cloudflareAiGatewayModelManagerOptions,
16
+ deepinfraModelManagerOptions,
16
17
  deepseekModelManagerOptions,
17
18
  firepassModelManagerOptions,
18
19
  fireworksModelManagerOptions,
@@ -127,7 +128,7 @@ function catalogDescriptor(
127
128
  * OpenAI code provider) are handled separately because they require different config shapes.
128
129
  */
129
130
  export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [
130
- descriptor("anthropic", "claude-sonnet-4-6", config => anthropicModelManagerOptions(config)),
131
+ descriptor("anthropic", "claude-sonnet-5", config => anthropicModelManagerOptions(config)),
131
132
  catalogDescriptor(
132
133
  "alibaba-coding-plan",
133
134
  "qwen3.5-plus",
@@ -168,6 +169,12 @@ export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [
168
169
  config => deepseekModelManagerOptions(config),
169
170
  catalog("DeepSeek", ["DEEPSEEK_API_KEY"]),
170
171
  ),
172
+ catalogDescriptor(
173
+ "deepinfra",
174
+ "deepseek-ai/DeepSeek-V3.2",
175
+ config => deepinfraModelManagerOptions(config),
176
+ catalog("DeepInfra", ["DEEPINFRA_API_KEY"]),
177
+ ),
171
178
  descriptor("mistral", "devstral-medium-latest", config => mistralModelManagerOptions(config)),
172
179
  catalogDescriptor(
173
180
  "nvidia",
@@ -665,6 +665,17 @@ export function deepseekModelManagerOptions(
665
665
  ): ModelManagerOptions<"openai-completions"> {
666
666
  return createSimpleOpenAICompletionsOptions("deepseek", "https://api.deepseek.com", config);
667
667
  }
668
+
669
+ export interface DeepInfraModelManagerConfig {
670
+ apiKey?: string;
671
+ baseUrl?: string;
672
+ }
673
+
674
+ export function deepinfraModelManagerOptions(
675
+ config?: DeepInfraModelManagerConfig,
676
+ ): ModelManagerOptions<"openai-completions"> {
677
+ return createSimpleOpenAICompletionsOptions("deepinfra", "https://api.deepinfra.com/v1/openai", config);
678
+ }
668
679
  // ---------------------------------------------------------------------------
669
680
  // 7.5 Fireworks
670
681
  // ---------------------------------------------------------------------------
@@ -1709,7 +1720,7 @@ export interface GithubCopilotModelManagerConfig {
1709
1720
  }
1710
1721
 
1711
1722
  function inferCopilotApi(modelId: string): Api {
1712
- if (/^claude-(haiku|sonnet|opus)-4([.-]|$)/.test(modelId)) {
1723
+ if (/^claude-(haiku|sonnet|opus)-(?:4|5)([.-]|$)/.test(modelId)) {
1713
1724
  return "anthropic-messages";
1714
1725
  }
1715
1726
  if (modelId.startsWith("gpt-5") || modelId.startsWith("oswe")) {
@@ -2138,7 +2149,7 @@ const COPILOT_DEFAULT_RESOLUTION = {
2138
2149
 
2139
2150
  const COPILOT_API_RESOLUTION_RULES: readonly ApiResolutionRule[] = [
2140
2151
  {
2141
- matches: modelId => /^claude-(haiku|sonnet|opus)-4([.-]|$)/.test(modelId),
2152
+ matches: modelId => /^claude-(haiku|sonnet|opus)-(?:4|5)([.-]|$)/.test(modelId),
2142
2153
  resolved: { api: "anthropic-messages", baseUrl: COPILOT_BASE_URL },
2143
2154
  },
2144
2155
  {
@@ -2282,6 +2293,8 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
2282
2293
  requiresAssistantContentForToolCalls: true,
2283
2294
  },
2284
2295
  }),
2296
+ // --- DeepInfra ---
2297
+ openAiCompletionsDescriptor("deepinfra", "deepinfra", "https://api.deepinfra.com/v1/openai"),
2285
2298
  ];
2286
2299
 
2287
2300
  const MODELS_DEV_PROVIDER_DESCRIPTORS_CODING_PLANS: readonly ModelsDevProviderDescriptor[] = [
@@ -21,6 +21,7 @@ import {
21
21
  } from "@gajae-code/utils";
22
22
  import { hasOpus47ApiRestrictions, mapEffortToAnthropicAdaptiveEffort } from "../model-thinking";
23
23
  import { calculateCost } from "../models";
24
+ import { isUsageLimitError } from "../rate-limit-utils";
24
25
  import { getEnvApiKey, OUTPUT_FALLBACK_BUFFER } from "../stream";
25
26
  import type {
26
27
  Api,
@@ -62,6 +63,7 @@ import { parseJsonWithRepair, parseStreamingJson } from "../utils/json-parse";
62
63
  import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
63
64
  import { notifyProviderResponse } from "../utils/provider-response";
64
65
  import { isCopilotTransientModelError } from "../utils/retry";
66
+ import { getRetryAfterMsFromHeaders } from "../utils/retry-after";
65
67
  import { resolveRetryBudget } from "../utils/retry-budget";
66
68
  import {
67
69
  COMBINATOR_KEYS,
@@ -724,6 +726,7 @@ export type AnthropicClientOptionsArgs = {
724
726
  onSseEvent?: AnthropicOptions["onSseEvent"];
725
727
  fetch?: FetchImpl;
726
728
  requestMaxRetries?: number;
729
+ maxRetryDelayMs?: number;
727
730
  };
728
731
 
729
732
  export type AnthropicClientOptionsResult = {
@@ -877,6 +880,101 @@ function mergeHeaders(...headerSources: (Record<string, string> | undefined)[]):
877
880
  return merged;
878
881
  }
879
882
 
883
+ const ANTHROPIC_RETRY_DELAY_CAP_MS = 60_000;
884
+ const ANTHROPIC_RATE_LIMIT_HEADER_PREFIX = "anthropic-ratelimit-";
885
+
886
+ function getSafeAnthropicHeaderEvidence(headers: Headers): string[] {
887
+ const evidence: string[] = [];
888
+ for (const [name, value] of headers) {
889
+ const lowerName = name.toLowerCase();
890
+ if (
891
+ lowerName === "retry-after" ||
892
+ lowerName === "retry-after-ms" ||
893
+ lowerName.startsWith(ANTHROPIC_RATE_LIMIT_HEADER_PREFIX)
894
+ ) {
895
+ evidence.push(`${lowerName}=${value}`);
896
+ }
897
+ }
898
+ return evidence.sort();
899
+ }
900
+
901
+ function getStringProperty(source: unknown, key: string): string | undefined {
902
+ if (!isRecord(source)) return undefined;
903
+ const value = source[key];
904
+ return typeof value === "string" ? value : undefined;
905
+ }
906
+
907
+ function appendAnthropicRateLimitEvidence(bodyText: string, headers: Headers): string {
908
+ const evidence = getSafeAnthropicHeaderEvidence(headers);
909
+ if (evidence.length === 0) return bodyText;
910
+
911
+ const suffix = ` Anthropic rate-limit evidence: ${evidence.join(", ")}`;
912
+ try {
913
+ const parsed = JSON.parse(bodyText) as unknown;
914
+ if (isRecord(parsed)) {
915
+ const error = parsed.error;
916
+ if (isRecord(error) && typeof error.message === "string" && !error.message.includes(suffix)) {
917
+ return JSON.stringify({ ...parsed, error: { ...error, message: `${error.message}${suffix}` } });
918
+ }
919
+ if (typeof parsed.message === "string" && !parsed.message.includes(suffix)) {
920
+ return JSON.stringify({ ...parsed, message: `${parsed.message}${suffix}` });
921
+ }
922
+ }
923
+ } catch {}
924
+
925
+ return bodyText.includes(suffix) ? bodyText : `${bodyText}${suffix}`;
926
+ }
927
+
928
+ function isAnthropicUsageExhaustionResponse(
929
+ bodyText: string,
930
+ headers: Headers,
931
+ retryAfterMs: number | undefined,
932
+ retryDelayCapMs: number,
933
+ ): boolean {
934
+ const overageReason = headers.get("anthropic-ratelimit-unified-overage-disabled-reason")?.toLowerCase();
935
+ if (overageReason === "out_of_credits") return true;
936
+ if (retryAfterMs !== undefined && retryAfterMs > retryDelayCapMs) return true;
937
+
938
+ try {
939
+ const parsed = JSON.parse(bodyText) as unknown;
940
+ const error = isRecord(parsed) ? parsed.error : undefined;
941
+ const type = getStringProperty(error, "type") ?? getStringProperty(parsed, "type");
942
+ const message = getStringProperty(error, "message") ?? getStringProperty(parsed, "message") ?? bodyText;
943
+ return (
944
+ /rate_limit_error/i.test(type ?? "") &&
945
+ (/request would exceed your account.?s rate limit/i.test(message) || /out_of_credits/i.test(message))
946
+ );
947
+ } catch {
948
+ return /request would exceed your account.?s rate limit|out_of_credits/i.test(bodyText);
949
+ }
950
+ }
951
+
952
+ function wrapAnthropicFetchForBoundedRateLimits(baseFetch: FetchImpl, maxRetryDelayMs: number | undefined): FetchImpl {
953
+ const retryDelayCapMs = maxRetryDelayMs ?? ANTHROPIC_RETRY_DELAY_CAP_MS;
954
+ return Object.assign(
955
+ async (input: string | URL | Request, init?: RequestInit): Promise<Response> => {
956
+ const response = await baseFetch(input, init);
957
+ if (response.status !== 429 || retryDelayCapMs === 0) return response;
958
+
959
+ const headers = new Headers(response.headers);
960
+ const retryAfterMs = getRetryAfterMsFromHeaders(headers);
961
+ const bodyText = await response
962
+ .clone()
963
+ .text()
964
+ .catch(() => "");
965
+ if (!isAnthropicUsageExhaustionResponse(bodyText, headers, retryAfterMs, retryDelayCapMs)) return response;
966
+
967
+ headers.set("x-should-retry", "false");
968
+ return new Response(appendAnthropicRateLimitEvidence(bodyText, headers), {
969
+ status: response.status,
970
+ statusText: response.statusText,
971
+ headers,
972
+ });
973
+ },
974
+ baseFetch.preconnect ? { preconnect: baseFetch.preconnect } : {},
975
+ );
976
+ }
977
+
880
978
  // The Anthropic SDK logs malformed SSE frames directly before rethrowing them.
881
979
  // We surface the resulting provider error ourselves, so keep the SDK quiet.
882
980
  const ANTHROPIC_SDK_LOG_LEVEL = "off" as const;
@@ -1042,6 +1140,7 @@ export function isProviderRetryableError(error: unknown, provider?: string): boo
1042
1140
  if (!(error instanceof Error)) return false;
1043
1141
  if (provider === "github-copilot" && isCopilotTransientModelError(error)) return true;
1044
1142
  const msg = error.message.toLowerCase();
1143
+ if (isUsageLimitError(error.message)) return false;
1045
1144
  if (
1046
1145
  isUnexpectedSocketCloseMessage(msg) ||
1047
1146
  /rate.?limit|too many requests|overloaded|service.?unavailable|internal_error|stream error.*received from peer|1302|timed?\s*out while waiting for the first event|timeout waiting for first/i.test(
@@ -1166,6 +1265,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1166
1265
  onSseEvent: options?.onSseEvent,
1167
1266
  fetch: options?.fetch,
1168
1267
  requestMaxRetries: options?.requestMaxRetries,
1268
+ maxRetryDelayMs: options?.maxRetryDelayMs,
1169
1269
  });
1170
1270
  client = created.client;
1171
1271
  isOAuthToken = created.isOAuthToken;
@@ -1723,11 +1823,8 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
1723
1823
  const foundryCustomHeaders = resolveAnthropicCustomHeaders(model);
1724
1824
  const tlsFetchOptions = buildClaudeCodeTlsFetchOptions(model, baseUrl);
1725
1825
  const baseFetch = args.fetch ?? fetch;
1726
- const debugFetch = onSseEvent
1727
- ? wrapFetchForSseDebug(baseFetch, event => onSseEvent(event, model))
1728
- : args.fetch
1729
- ? baseFetch
1730
- : undefined;
1826
+ const boundedFetch = wrapAnthropicFetchForBoundedRateLimits(baseFetch, args.maxRetryDelayMs);
1827
+ const debugFetch = onSseEvent ? wrapFetchForSseDebug(boundedFetch, event => onSseEvent(event, model)) : boundedFetch;
1731
1828
  if (model.provider === "github-copilot") {
1732
1829
  const copilotApiKey = parseGitHubCopilotApiKey(apiKey).accessToken;
1733
1830
  const betaFeatures = [...extraBetas];
@@ -1755,7 +1852,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
1755
1852
  dangerouslyAllowBrowser: true,
1756
1853
  defaultHeaders,
1757
1854
  logLevel: ANTHROPIC_SDK_LOG_LEVEL,
1758
- ...(debugFetch ? { fetch: debugFetch } : {}),
1855
+ fetch: debugFetch,
1759
1856
  ...(tlsFetchOptions ? { fetchOptions: tlsFetchOptions } : {}),
1760
1857
  };
1761
1858
  }
@@ -1789,7 +1886,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
1789
1886
  dangerouslyAllowBrowser: true,
1790
1887
  defaultHeaders,
1791
1888
  logLevel: ANTHROPIC_SDK_LOG_LEVEL,
1792
- ...(debugFetch ? { fetch: debugFetch } : {}),
1889
+ fetch: debugFetch,
1793
1890
  };
1794
1891
  }
1795
1892
 
@@ -1802,7 +1899,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
1802
1899
  dangerouslyAllowBrowser: true,
1803
1900
  defaultHeaders,
1804
1901
  logLevel: ANTHROPIC_SDK_LOG_LEVEL,
1805
- ...(debugFetch ? { fetch: debugFetch } : {}),
1902
+ fetch: debugFetch,
1806
1903
  ...(tlsFetchOptions ? { fetchOptions: tlsFetchOptions } : {}),
1807
1904
  };
1808
1905
  }
@@ -3,8 +3,13 @@
3
3
  * Uses the same format as the official Gemini CLI (v0.35+):
4
4
  * GeminiCLI/VERSION/MODEL (PLATFORM; ARCH; SURFACE)
5
5
  */
6
+ export const GEMINI_CLI_VERSION_ENV = "GJC_AI_GEMINI_CLI_VERSION";
7
+ export const LEGACY_GEMINI_CLI_VERSION_ENV = "PI_AI_GEMINI_CLI_VERSION";
8
+ export const DEFAULT_GEMINI_CLI_VERSION = "0.49.0";
9
+
6
10
  export function getGeminiCliUserAgent(modelId = "gemini-3.1-pro-preview"): string {
7
- const version = process.env.PI_AI_GEMINI_CLI_VERSION || "0.46.0";
11
+ const version =
12
+ process.env[GEMINI_CLI_VERSION_ENV] || process.env[LEGACY_GEMINI_CLI_VERSION_ENV] || DEFAULT_GEMINI_CLI_VERSION;
8
13
  const platform = process.platform === "win32" ? "win32" : process.platform;
9
14
  const arch = process.arch === "x64" ? "x64" : process.arch;
10
15
  return `GeminiCLI/${version}/${modelId} (${platform}; ${arch}; terminal)`;
@@ -1355,9 +1355,12 @@ export function parseChunkUsage(
1355
1355
  ): AssistantMessage["usage"] {
1356
1356
  const promptTokenDetails = getOptionalObjectProperty(rawUsage, "prompt_tokens_details");
1357
1357
  const completionTokenDetails = getOptionalObjectProperty(rawUsage, "completion_tokens_details");
1358
+ const deepSeekCacheHitTokens = getOptionalNumberProperty(rawUsage, "prompt_cache_hit_tokens");
1359
+ const deepSeekCacheMissTokens = getOptionalNumberProperty(rawUsage, "prompt_cache_miss_tokens");
1358
1360
  const cachedTokens =
1359
1361
  getOptionalNumberProperty(rawUsage, "cached_tokens") ??
1360
1362
  (promptTokenDetails ? getOptionalNumberProperty(promptTokenDetails, "cached_tokens") : undefined) ??
1363
+ deepSeekCacheHitTokens ??
1361
1364
  0;
1362
1365
  // OpenRouter exposes cache writes via `prompt_tokens_details.cache_write_tokens`
1363
1366
  // and INCLUDES them in `prompt_tokens`. Without subtracting, cache-write tokens
@@ -1369,7 +1372,7 @@ export function parseChunkUsage(
1369
1372
  const reasoningTokens =
1370
1373
  (completionTokenDetails ? getOptionalNumberProperty(completionTokenDetails, "reasoning_tokens") : undefined) ?? 0;
1371
1374
  const promptTokens = getOptionalNumberProperty(rawUsage, "prompt_tokens") ?? 0;
1372
- const input = Math.max(0, promptTokens - cachedTokens - cacheWriteTokens);
1375
+ const input = Math.max(0, deepSeekCacheMissTokens ?? promptTokens - cachedTokens - cacheWriteTokens);
1373
1376
  // Per OpenAI's CompletionUsage spec, `reasoning_tokens` is a subset of
1374
1377
  // `completion_tokens` (which is the total billed output). Adding them would
1375
1378
  // double-count.
@@ -36,6 +36,14 @@ export function parseRateLimitReason(errorMessage: string): RateLimitReason {
36
36
  return "MODEL_CAPACITY_EXHAUSTED";
37
37
  }
38
38
 
39
+ if (
40
+ lower.includes("out_of_credits") ||
41
+ lower.includes("request would exceed your account's rate limit") ||
42
+ lower.includes("request would exceed your accounts rate limit")
43
+ ) {
44
+ return "QUOTA_EXHAUSTED";
45
+ }
46
+
39
47
  if (
40
48
  lower.includes("per minute") ||
41
49
  lower.includes("rate limit") ||
@@ -86,8 +94,7 @@ export function calculateRateLimitBackoffMs(reason: RateLimitReason): number {
86
94
 
87
95
  /** Detect usage/quota limit errors in error messages (persistent, requires credential switch). */
88
96
  const USAGE_LIMIT_PATTERN =
89
- /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|model.?limit|model_limit_reached|message.?limit|message_limit_reached|limit for this model|quota.?exceeded|resource has been exhausted[^\n]*(?:quota|limit)/i;
90
-
97
+ /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|model.?limit|model_limit_reached|message.?limit|message_limit_reached|limit for this model|quota.?exceeded|out_of_credits|request would exceed your account.?s rate limit|resource has been exhausted[^\n]*(?:quota|limit)/i;
91
98
  export function isUsageLimitError(errorMessage: string): boolean {
92
99
  return USAGE_LIMIT_PATTERN.test(errorMessage);
93
100
  }
package/src/stream.ts CHANGED
@@ -98,6 +98,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
98
98
  "opencode-zen": "OPENCODE_API_KEY",
99
99
  cursor: "CURSOR_ACCESS_TOKEN",
100
100
  deepseek: "DEEPSEEK_API_KEY",
101
+ deepinfra: "DEEPINFRA_API_KEY",
101
102
  "openai-codex": "OPENAI_CODEX_OAUTH_TOKEN",
102
103
  "azure-openai": "AZURE_OPENAI_API_KEY",
103
104
  "azure-openai-responses": "AZURE_OPENAI_API_KEY",
package/src/types.ts CHANGED
@@ -116,6 +116,7 @@ export type KnownProvider =
116
116
  | "gitlab-duo"
117
117
  | "cursor"
118
118
  | "deepseek"
119
+ | "deepinfra"
119
120
  | "xai"
120
121
  | "groq"
121
122
  | "cerebras"
@@ -186,7 +187,7 @@ export type CacheRetention = "none" | "short" | "long";
186
187
  *
187
188
  * The unscoped values (`"auto"`, `"default"`, `"flex"`, `"scale"`,
188
189
  * `"priority"`) are passed through to providers that understand them
189
- * (OpenAI's `service_tier` field directly; Anthropic translates
190
+ * (OpenAI and DeepInfra's `service_tier` field directly; Anthropic translates
190
191
  * `"priority"` into `speed: "fast"` on supported Opus models).
191
192
  *
192
193
  * The scoped values target a specific provider family and behave as the
@@ -223,17 +224,17 @@ export function resolveServiceTier(
223
224
  }
224
225
 
225
226
  /**
226
- * True when the (possibly scoped) tier should be sent as OpenAI's
227
- * `service_tier` request field for the given provider. Non-OpenAI
228
- * providers, unsupported tiers (`"auto"`, `"default"`), and scope
229
- * mismatches all return false.
227
+ * True when the (possibly scoped) tier should be sent as an OpenAI-compatible
228
+ * `service_tier` request field for providers that support it. Unsupported tiers
229
+ * (`"auto"`, `"default"`) and scope mismatches all return false.
230
230
  */
231
231
  export function shouldSendServiceTier(
232
232
  serviceTier: ServiceTier | null | undefined,
233
233
  provider: Provider | undefined,
234
234
  ): boolean {
235
- if (provider !== "openai" && provider !== "openai-codex") return false;
236
235
  const resolved = resolveServiceTier(serviceTier, provider);
236
+ if (provider === "deepinfra") return resolved === "priority";
237
+ if (provider !== "openai" && provider !== "openai-codex") return false;
237
238
  return resolved === "flex" || resolved === "scale" || resolved === "priority";
238
239
  }
239
240
 
@@ -252,7 +253,9 @@ export function getPriorityPremiumRequests(
252
253
  if (resolveServiceTier(serviceTier, provider) !== "priority") return 0;
253
254
  // Only providers that realize `priority` on the wire bill the user.
254
255
  // Everywhere else, the field is silently dropped and nothing is charged.
255
- return provider === "openai" || provider === "openai-codex" || provider === "anthropic" ? 1 : 0;
256
+ return provider === "openai" || provider === "openai-codex" || provider === "anthropic" || provider === "deepinfra"
257
+ ? 1
258
+ : 0;
256
259
  }
257
260
 
258
261
  export interface ProviderSessionState {
@@ -0,0 +1,15 @@
1
+ /** DeepInfra login flow (API key paste against https://api.deepinfra.com/v1/openai). */
2
+ import { createApiKeyLogin } from "./api-key-login";
3
+
4
+ export const loginDeepInfra = createApiKeyLogin({
5
+ providerLabel: "DeepInfra",
6
+ authUrl: "https://deepinfra.com/dash/api_keys",
7
+ instructions: "Create or copy your DeepInfra API key from the DeepInfra dashboard",
8
+ promptMessage: "Paste your DeepInfra API key",
9
+ placeholder: "sk-...",
10
+ validation: {
11
+ kind: "models-endpoint",
12
+ provider: "DeepInfra",
13
+ modelsUrl: "https://api.deepinfra.com/v1/openai/models",
14
+ },
15
+ });
@@ -60,6 +60,11 @@ const builtInOAuthProviders: OAuthProviderInfo[] = [
60
60
  name: "DeepSeek",
61
61
  available: true,
62
62
  },
63
+ {
64
+ id: "deepinfra",
65
+ name: "DeepInfra",
66
+ available: true,
67
+ },
63
68
  {
64
69
  id: "xai",
65
70
  name: "xAI",
@@ -356,6 +361,7 @@ export async function refreshOAuthToken(
356
361
  case "opencode-zen":
357
362
  case "opencode-go":
358
363
  case "cerebras":
364
+ case "deepinfra":
359
365
  case "fireworks":
360
366
  case "firepass":
361
367
  case "fugu":
@@ -15,6 +15,7 @@ export type OAuthProvider =
15
15
  | "cloudflare-ai-gateway"
16
16
  | "cursor"
17
17
  | "deepseek"
18
+ | "deepinfra"
18
19
  | "fireworks"
19
20
  | "firepass"
20
21
  | "fugu"