@gajae-code/ai 0.11.8 → 0.11.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,12 +2,24 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.11.10] - 2026-07-25
6
+
7
+ ## [0.11.9] - 2026-07-24
8
+ ### Fixed
9
+
10
+ - Credential selection and aggregate usage callers now stop awaiting immediately when their own signal aborts without cancelling shared usage fetches, and ranking deadlines no longer re-await the same stalled usage request during credential resolution.
11
+ - Kimi Code now allows one continuous 300-second first-event wait before aborting, while preserving explicit caller and environment timeout overrides and the existing inter-event idle timeout.
12
+
13
+ ### Added
14
+
15
+ - Added first-class support for **OpenGateway by Sionic AI**, an OpenAI-compatible gateway. Registers the `opengateway` provider descriptor, `/login` OAuth entry (API-key paste validated against `https://apis.opengateway.ai/v1/models`), `OPENGATEWAY_API_KEY` environment resolution, and bundled `models.json` seed models. Models are discovered dynamically from the OpenAI-compatible `/v1/models` endpoint (base URL `https://apis.opengateway.ai/v1`).
16
+
5
17
  ## [0.11.8] - 2026-07-23
6
18
 
7
19
  ### Fixed
8
20
 
9
21
  - OpenAI Responses / Codex native history replay no longer submits missing resident-image placeholders as `input_image.image_url`. Invalid values (including `[Session resident imageUrl blob missing: …]`) are dropped, or retained as `file_id`-only parts when a non-empty `file_id` is present, so a single unavailable historical image cannot brick `/retry` (#2924).
10
- - Raised the first-event stream timeout floor to five minutes for `alibaba-token-plan` models using the OpenAI Completions and Responses APIs, while preserving caller and environment overrides and the existing inter-event idle timeout.
22
+ - Raised the first-event stream timeout floor to five minutes for `alibaba-token-plan` models at both the OpenAI provider and outer lazy-stream watchdogs, while preserving caller and environment overrides and the existing inter-event idle timeout.
11
23
  - OAuth refresh peer-rotation recovery now runs before failure classification instead of only on the definitive-failure path, and the definitive matcher recognizes the "grant is invalid" phrasing. Providers whose invalid-grant response does not contain the literal `invalid_grant` (e.g. Kimi's 400 "The provided authorization grant is invalid") previously had rotation races misclassified as transient, temp-blocking a healthy credential for five minutes on every race; with Kimi's ~12-minute access tokens and multiple processes sharing the credential store this surfaced as repeated logouts. Genuine revocations are now disabled with a cause instead of looping temp-blocks.
12
24
  - Anthropic 400 `Invalid \`signature\` in \`thinking\` block` responses now trigger the one-shot thinking replay repair instead of failing the turn. The existing repair matcher only recognized the "latest assistant message ... cannot be modified" wording, so the signature-validation variant — which can cite a `thinking`/`redacted_thinking` block anywhere in the replayed history (e.g. after compaction/pruning rewrote an earlier turn) — was treated as a fatal request error. The retry now rebuilds the request with thinking blocks dropped from every replayed assistant message (`repairAllAssistantThinking`), while the latest-message mutation variant keeps the targeted latest-only repair.
13
25
 
package/README.md CHANGED
@@ -69,6 +69,7 @@ Unified LLM API with automatic model discovery, provider configuration, token an
69
69
  - **MiniMax Coding Plan** (requires `MINIMAX_CODE_API_KEY` or `MINIMAX_CODE_CN_API_KEY`)
70
70
  - **Xiaomi MiMo** (requires `XIAOMI_API_KEY`)
71
71
  - **ZenMux** (requires `ZENMUX_API_KEY`)
72
+ - **OpenGateway by Sionic AI** (requires `OPENGATEWAY_API_KEY`)
72
73
  - **Qwen Portal** (supports `QWEN_OAUTH_TOKEN` or `QWEN_PORTAL_API_KEY`)
73
74
  - **Cloudflare AI Gateway** (requires `CLOUDFLARE_AI_GATEWAY_API_KEY` and provider-specific gateway base URL)
74
75
  - **Ollama** (local OpenAI-compatible runtime; optional `OLLAMA_API_KEY`)
@@ -954,6 +955,7 @@ In Node.js environments, you can set environment variables to avoid passing API
954
955
  | MiniMax Code | `MINIMAX_CODE_API_KEY` (international) or `MINIMAX_CODE_CN_API_KEY` (China) |
955
956
  | Xiaomi MiMo | `XIAOMI_API_KEY` |
956
957
  | ZenMux | `ZENMUX_API_KEY` |
958
+ | OpenGateway | `OPENGATEWAY_API_KEY` |
957
959
  | vLLM | `VLLM_API_KEY` |
958
960
  | Cloudflare AI Gateway | `CLOUDFLARE_AI_GATEWAY_API_KEY` |
959
961
  | GitHub Copilot | `COPILOT_GITHUB_TOKEN` or `GH_TOKEN` or `GITHUB_TOKEN` |
@@ -977,6 +979,7 @@ Provider endpoint defaults for the current OpenAI-compatible integrations:
977
979
  - Xiaomi MiMo: `https://api.xiaomimimo.com/anthropic`
978
980
  - ZenMux (OpenAI): `https://zenmux.ai/api/v1`
979
981
  - ZenMux (Anthropic models): `https://zenmux.ai/api/anthropic`
982
+ - OpenGateway by Sionic AI: `https://apis.opengateway.ai/v1`
980
983
  - vLLM: `http://127.0.0.1:8000/v1`
981
984
  - Ollama: local OpenAI-compatible runtime (`http://127.0.0.1:11434/v1`)
982
985
  - Ollama Cloud: native Ollama API host (`https://ollama.com/api`, configured here as base URL `https://ollama.com`)
@@ -111,6 +111,16 @@ export interface ZenMuxModelManagerConfig {
111
111
  baseUrl?: string;
112
112
  }
113
113
  export declare function zenmuxModelManagerOptions(config?: ZenMuxModelManagerConfig): ModelManagerOptions<Api>;
114
+ export interface OpenGatewayModelManagerConfig {
115
+ apiKey?: string;
116
+ baseUrl?: string;
117
+ }
118
+ /**
119
+ * OpenGateway by Sionic AI — an OpenAI-compatible gateway that fronts OpenAI,
120
+ * Anthropic, and Google models behind one API key. Models are discovered from
121
+ * the OpenAI-compatible `/v1/models` endpoint.
122
+ */
123
+ export declare function opengatewayModelManagerOptions(config?: OpenGatewayModelManagerConfig): ModelManagerOptions<"openai-completions">;
114
124
  export interface KiloModelManagerConfig {
115
125
  apiKey?: string;
116
126
  baseUrl?: string;
@@ -17,6 +17,14 @@ interface BedrockProviderModule {
17
17
  streamBedrock: (model: Model<"bedrock-converse-stream">, context: Context, options: BedrockOptions) => AssistantMessageEventStream;
18
18
  }
19
19
  export declare function setBedrockProviderModule(module: BedrockProviderModule): void;
20
+ /**
21
+ * Resolves the first-event timeout fallback for the outer lazy-stream watchdog.
22
+ * A configured wrapper-specific fallback (from `LazyStreamLimits`) always wins;
23
+ * otherwise providers known to have slow first events get a five-minute floor
24
+ * matching their inner provider-level override. Returns `undefined` for
25
+ * providers that should use the shared default.
26
+ */
27
+ export declare function resolveLazyStreamFirstEventFallbackMs(provider: string, configuredFallbackMs?: number): number | undefined;
20
28
  export declare const streamAnthropic: (model: Model<"anthropic-messages">, context: Context, options: OptionsForApi<"anthropic-messages">) => EventStreamImpl;
21
29
  export declare const streamAzureOpenAIResponses: (model: Model<"azure-openai-responses">, context: Context, options: OptionsForApi<"azure-openai-responses">) => EventStreamImpl;
22
30
  export declare const streamGoogle: (model: Model<"google-generative-ai">, context: Context, options: OptionsForApi<"google-generative-ai">) => EventStreamImpl;
@@ -51,7 +51,7 @@ export interface ThinkingConfig {
51
51
  /** Provider-specific transport used to encode the selected effort. */
52
52
  mode: ThinkingControlMode;
53
53
  }
54
- export type KnownProvider = "alibaba-token-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
54
+ export type KnownProvider = "alibaba-token-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "opengateway" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
55
55
  export type Provider = KnownProvider | string;
56
56
  import type { Effort } from "./model-thinking";
57
57
  /** Token budgets for each thinking level (token-based providers only) */
@@ -1,3 +1,4 @@
1
+ export declare function getProviderFirstEventTimeoutFallbackMs(provider: string): number | undefined;
1
2
  /**
2
3
  * Returns the idle timeout used for provider streaming transports.
3
4
  *
@@ -0,0 +1 @@
1
+ export declare const loginOpenGateway: (options: import("./types").OAuthController) => Promise<string>;
@@ -7,7 +7,7 @@ export type OAuthCredentials = {
7
7
  email?: string;
8
8
  accountId?: string;
9
9
  };
10
- export type OAuthProvider = "alibaba-token-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
10
+ export type OAuthProvider = "alibaba-token-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "opengateway" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
11
11
  export type OAuthProviderId = OAuthProvider | (string & {});
12
12
  export type OAuthPrompt = {
13
13
  message: string;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/ai",
4
- "version": "0.11.8",
4
+ "version": "0.11.10",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://gajae-code.com",
7
7
  "author": "Yeachan-Heo and Gajae Code Contributors",
@@ -40,7 +40,7 @@
40
40
  "dependencies": {
41
41
  "@anthropic-ai/sdk": "^0.94.0",
42
42
  "@bufbuild/protobuf": "^2.12.0",
43
- "@gajae-code/utils": "0.11.8",
43
+ "@gajae-code/utils": "0.11.10",
44
44
  "openai": "^6.36.0",
45
45
  "partial-json": "^0.1.7",
46
46
  "zod": "4.4.3"
@@ -1998,6 +1998,12 @@ export class AuthStorage {
1998
1998
  await saveApiKeyCredential(apiKey);
1999
1999
  return;
2000
2000
  }
2001
+ case "opengateway": {
2002
+ const { loginOpenGateway } = await import("./utils/oauth/opengateway");
2003
+ const apiKey = await loginOpenGateway(ctrl);
2004
+ await saveApiKeyCredential(apiKey);
2005
+ return;
2006
+ }
2001
2007
  default: {
2002
2008
  const customProvider = getOAuthProvider(provider);
2003
2009
  if (!customProvider) {
@@ -2508,9 +2514,12 @@ export class AuthStorage {
2508
2514
  if (storeHook) {
2509
2515
  return storeHook(provider, credential, options?.signal);
2510
2516
  }
2511
- return this.#fetchUsageCached(
2512
- this.#buildUsageRequestForOauth(provider, credential, options?.baseUrl),
2513
- options?.timeoutMs ?? this.#usageRequestTimeoutMs,
2517
+ return raceUsageWithSignal(
2518
+ this.#fetchUsageCached(
2519
+ this.#buildUsageRequestForOauth(provider, credential, options?.baseUrl),
2520
+ options?.timeoutMs ?? this.#usageRequestTimeoutMs,
2521
+ ),
2522
+ options?.signal,
2514
2523
  );
2515
2524
  }
2516
2525
 
@@ -2562,7 +2571,7 @@ export class AuthStorage {
2562
2571
  const cacheKey = this.#buildUsageReportsCacheKey(requests);
2563
2572
 
2564
2573
  const inFlight = this.#usageReportsInFlight.get(cacheKey);
2565
- if (inFlight) return inFlight;
2574
+ if (inFlight) return raceUsageWithSignal(inFlight, options?.signal);
2566
2575
 
2567
2576
  const promise = (async () => {
2568
2577
  if (options?.logDetails !== false) {
@@ -2610,7 +2619,7 @@ export class AuthStorage {
2610
2619
  });
2611
2620
 
2612
2621
  this.#usageReportsInFlight.set(cacheKey, promise);
2613
- return promise;
2622
+ return raceUsageWithSignal(promise, options?.signal);
2614
2623
  }
2615
2624
 
2616
2625
  /**
@@ -2861,7 +2870,7 @@ export class AuthStorage {
2861
2870
  const blockedUntil = this.#getCredentialBlockedUntil(args.providerKey, selection.index);
2862
2871
  if (blockedUntil !== undefined) return { selection, usage: null, usageChecked: false, blockedUntil };
2863
2872
  const usage = await this.#getUsageReport(args.provider, selection.credential, {
2864
- ...args.options,
2873
+ baseUrl: args.options?.baseUrl,
2865
2874
  timeoutMs: this.#usageRequestTimeoutMs,
2866
2875
  });
2867
2876
  return { selection, usage, usageChecked: true, blockedUntil: undefined as number | undefined };
@@ -2874,16 +2883,21 @@ export class AuthStorage {
2874
2883
  // path so memory drops immediately.
2875
2884
  const timer = setTimeout(() => timeoutSignal.resolve(null), usageTimeout);
2876
2885
  timer.unref?.();
2877
- const usageResults = await Promise.race([usagePromise, timeoutSignal.promise]).then(result => {
2878
- clearTimeout(timer);
2879
- return (
2880
- result ??
2881
- args.order.map(idx => {
2882
- const selection = args.credentials[idx];
2883
- return selection ? { selection, usage: null, usageChecked: false, blockedUntil: undefined } : null;
2884
- })
2886
+ let resolvedUsageResults: Awaited<typeof usagePromise> | null;
2887
+ try {
2888
+ resolvedUsageResults = await raceUsageWithSignal(
2889
+ Promise.race([usagePromise, timeoutSignal.promise]),
2890
+ args.options?.signal,
2885
2891
  );
2886
- });
2892
+ } finally {
2893
+ clearTimeout(timer);
2894
+ }
2895
+ const usageResults =
2896
+ resolvedUsageResults ??
2897
+ args.order.map(idx => {
2898
+ const selection = args.credentials[idx];
2899
+ return selection ? { selection, usage: null, usageChecked: true, blockedUntil: undefined } : null;
2900
+ });
2887
2901
 
2888
2902
  for (let orderPos = 0; orderPos < usageResults.length; orderPos += 1) {
2889
2903
  const result = usageResults[orderPos];
package/src/cli.ts CHANGED
@@ -118,6 +118,7 @@ Providers:
118
118
  minimax-code-cn MiniMax Coding Plan (China)
119
119
  cursor Cursor (Anthropic, GPT, etc.)
120
120
  zenmux ZenMux
121
+ opengateway OpenGateway by Sionic AI
121
122
  ollama-cloud Ollama Cloud
122
123
 
123
124
  Examples:
package/src/models.json CHANGED
@@ -7,12 +7,25 @@
7
7
  "provider": "alibaba-token-plan",
8
8
  "baseUrl": "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1",
9
9
  "reasoning": true,
10
- "input": ["text"],
11
- "cost": { "input": 0, "output": 0, "cacheRead": 0, "cacheWrite": 0 },
10
+ "input": [
11
+ "text"
12
+ ],
13
+ "cost": {
14
+ "input": 0,
15
+ "output": 0,
16
+ "cacheRead": 0,
17
+ "cacheWrite": 0
18
+ },
12
19
  "contextWindow": 1000000,
13
20
  "maxTokens": 384000,
14
- "compat": { "supportsDeveloperRole": false },
15
- "thinking": { "mode": "effort", "minLevel": "minimal", "maxLevel": "xhigh" }
21
+ "compat": {
22
+ "supportsDeveloperRole": false
23
+ },
24
+ "thinking": {
25
+ "mode": "effort",
26
+ "minLevel": "minimal",
27
+ "maxLevel": "xhigh"
28
+ }
16
29
  },
17
30
  "glm-5.2": {
18
31
  "id": "glm-5.2",
@@ -21,12 +34,25 @@
21
34
  "provider": "alibaba-token-plan",
22
35
  "baseUrl": "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1",
23
36
  "reasoning": true,
24
- "input": ["text"],
25
- "cost": { "input": 0, "output": 0, "cacheRead": 0, "cacheWrite": 0 },
37
+ "input": [
38
+ "text"
39
+ ],
40
+ "cost": {
41
+ "input": 0,
42
+ "output": 0,
43
+ "cacheRead": 0,
44
+ "cacheWrite": 0
45
+ },
26
46
  "contextWindow": 1000000,
27
47
  "maxTokens": 131072,
28
- "compat": { "supportsDeveloperRole": false },
29
- "thinking": { "mode": "effort", "minLevel": "minimal", "maxLevel": "xhigh" }
48
+ "compat": {
49
+ "supportsDeveloperRole": false
50
+ },
51
+ "thinking": {
52
+ "mode": "effort",
53
+ "minLevel": "minimal",
54
+ "maxLevel": "xhigh"
55
+ }
30
56
  },
31
57
  "qwen3.8-max-preview": {
32
58
  "id": "qwen3.8-max-preview",
@@ -35,12 +61,25 @@
35
61
  "provider": "alibaba-token-plan",
36
62
  "baseUrl": "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1",
37
63
  "reasoning": true,
38
- "input": ["text"],
39
- "cost": { "input": 0, "output": 0, "cacheRead": 0, "cacheWrite": 0 },
64
+ "input": [
65
+ "text"
66
+ ],
67
+ "cost": {
68
+ "input": 0,
69
+ "output": 0,
70
+ "cacheRead": 0,
71
+ "cacheWrite": 0
72
+ },
40
73
  "contextWindow": 1000000,
41
74
  "maxTokens": 65536,
42
- "compat": { "supportsDeveloperRole": false },
43
- "thinking": { "mode": "effort", "minLevel": "minimal", "maxLevel": "xhigh" }
75
+ "compat": {
76
+ "supportsDeveloperRole": false
77
+ },
78
+ "thinking": {
79
+ "mode": "effort",
80
+ "minLevel": "minimal",
81
+ "maxLevel": "xhigh"
82
+ }
44
83
  }
45
84
  },
46
85
  "amazon-bedrock": {
@@ -3142,6 +3181,198 @@
3142
3181
  "minLevel": "minimal",
3143
3182
  "maxLevel": "high"
3144
3183
  }
3184
+ },
3185
+ "anthropic.claude-opus-5": {
3186
+ "id": "anthropic.claude-opus-5",
3187
+ "name": "Anthropic Opus 5",
3188
+ "api": "bedrock-converse-stream",
3189
+ "provider": "amazon-bedrock",
3190
+ "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
3191
+ "reasoning": true,
3192
+ "input": [
3193
+ "text",
3194
+ "image"
3195
+ ],
3196
+ "cost": {
3197
+ "input": 5,
3198
+ "output": 25,
3199
+ "cacheRead": 0.5,
3200
+ "cacheWrite": 6.25
3201
+ },
3202
+ "contextWindow": 1000000,
3203
+ "maxTokens": 128000,
3204
+ "thinking": {
3205
+ "mode": "anthropic-adaptive",
3206
+ "minLevel": "minimal",
3207
+ "maxLevel": "max",
3208
+ "levels": [
3209
+ "minimal",
3210
+ "low",
3211
+ "medium",
3212
+ "high",
3213
+ "max"
3214
+ ]
3215
+ }
3216
+ },
3217
+ "au.anthropic.claude-opus-5": {
3218
+ "id": "au.anthropic.claude-opus-5",
3219
+ "name": "Anthropic Opus 5 (AU)",
3220
+ "api": "bedrock-converse-stream",
3221
+ "provider": "amazon-bedrock",
3222
+ "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
3223
+ "reasoning": true,
3224
+ "input": [
3225
+ "text",
3226
+ "image"
3227
+ ],
3228
+ "cost": {
3229
+ "input": 5,
3230
+ "output": 25,
3231
+ "cacheRead": 0.5,
3232
+ "cacheWrite": 6.25
3233
+ },
3234
+ "contextWindow": 1000000,
3235
+ "maxTokens": 128000,
3236
+ "thinking": {
3237
+ "mode": "anthropic-adaptive",
3238
+ "minLevel": "minimal",
3239
+ "maxLevel": "max",
3240
+ "levels": [
3241
+ "minimal",
3242
+ "low",
3243
+ "medium",
3244
+ "high",
3245
+ "max"
3246
+ ]
3247
+ }
3248
+ },
3249
+ "eu.anthropic.claude-opus-5": {
3250
+ "id": "eu.anthropic.claude-opus-5",
3251
+ "name": "Anthropic Opus 5 (EU)",
3252
+ "api": "bedrock-converse-stream",
3253
+ "provider": "amazon-bedrock",
3254
+ "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
3255
+ "reasoning": true,
3256
+ "input": [
3257
+ "text",
3258
+ "image"
3259
+ ],
3260
+ "cost": {
3261
+ "input": 5.5,
3262
+ "output": 27.5,
3263
+ "cacheRead": 0.55,
3264
+ "cacheWrite": 6.875
3265
+ },
3266
+ "contextWindow": 1000000,
3267
+ "maxTokens": 128000,
3268
+ "thinking": {
3269
+ "mode": "anthropic-adaptive",
3270
+ "minLevel": "minimal",
3271
+ "maxLevel": "max",
3272
+ "levels": [
3273
+ "minimal",
3274
+ "low",
3275
+ "medium",
3276
+ "high",
3277
+ "max"
3278
+ ]
3279
+ }
3280
+ },
3281
+ "global.anthropic.claude-opus-5": {
3282
+ "id": "global.anthropic.claude-opus-5",
3283
+ "name": "Anthropic Opus 5 (Global)",
3284
+ "api": "bedrock-converse-stream",
3285
+ "provider": "amazon-bedrock",
3286
+ "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
3287
+ "reasoning": true,
3288
+ "input": [
3289
+ "text",
3290
+ "image"
3291
+ ],
3292
+ "cost": {
3293
+ "input": 5,
3294
+ "output": 25,
3295
+ "cacheRead": 0.5,
3296
+ "cacheWrite": 6.25
3297
+ },
3298
+ "contextWindow": 1000000,
3299
+ "maxTokens": 128000,
3300
+ "thinking": {
3301
+ "mode": "anthropic-adaptive",
3302
+ "minLevel": "minimal",
3303
+ "maxLevel": "max",
3304
+ "levels": [
3305
+ "minimal",
3306
+ "low",
3307
+ "medium",
3308
+ "high",
3309
+ "max"
3310
+ ]
3311
+ }
3312
+ },
3313
+ "jp.anthropic.claude-opus-5": {
3314
+ "id": "jp.anthropic.claude-opus-5",
3315
+ "name": "Anthropic Opus 5 (JP)",
3316
+ "api": "bedrock-converse-stream",
3317
+ "provider": "amazon-bedrock",
3318
+ "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
3319
+ "reasoning": true,
3320
+ "input": [
3321
+ "text",
3322
+ "image"
3323
+ ],
3324
+ "cost": {
3325
+ "input": 5,
3326
+ "output": 25,
3327
+ "cacheRead": 0.5,
3328
+ "cacheWrite": 6.25
3329
+ },
3330
+ "contextWindow": 1000000,
3331
+ "maxTokens": 128000,
3332
+ "thinking": {
3333
+ "mode": "anthropic-adaptive",
3334
+ "minLevel": "minimal",
3335
+ "maxLevel": "max",
3336
+ "levels": [
3337
+ "minimal",
3338
+ "low",
3339
+ "medium",
3340
+ "high",
3341
+ "max"
3342
+ ]
3343
+ }
3344
+ },
3345
+ "us.anthropic.claude-opus-5": {
3346
+ "id": "us.anthropic.claude-opus-5",
3347
+ "name": "Anthropic Opus 5 (US)",
3348
+ "api": "bedrock-converse-stream",
3349
+ "provider": "amazon-bedrock",
3350
+ "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
3351
+ "reasoning": true,
3352
+ "input": [
3353
+ "text",
3354
+ "image"
3355
+ ],
3356
+ "cost": {
3357
+ "input": 5,
3358
+ "output": 25,
3359
+ "cacheRead": 0.5,
3360
+ "cacheWrite": 6.25
3361
+ },
3362
+ "contextWindow": 1000000,
3363
+ "maxTokens": 128000,
3364
+ "thinking": {
3365
+ "mode": "anthropic-adaptive",
3366
+ "minLevel": "minimal",
3367
+ "maxLevel": "max",
3368
+ "levels": [
3369
+ "minimal",
3370
+ "low",
3371
+ "medium",
3372
+ "high",
3373
+ "max"
3374
+ ]
3375
+ }
3145
3376
  }
3146
3377
  },
3147
3378
  "anthropic": {
@@ -3661,6 +3892,31 @@
3661
3892
  "minLevel": "minimal",
3662
3893
  "maxLevel": "high"
3663
3894
  }
3895
+ },
3896
+ "claude-opus-5": {
3897
+ "id": "claude-opus-5",
3898
+ "name": "Anthropic Opus 5",
3899
+ "api": "anthropic-messages",
3900
+ "provider": "anthropic",
3901
+ "baseUrl": "https://api.anthropic.com",
3902
+ "reasoning": true,
3903
+ "input": [
3904
+ "text",
3905
+ "image"
3906
+ ],
3907
+ "cost": {
3908
+ "input": 5,
3909
+ "output": 25,
3910
+ "cacheRead": 0.5,
3911
+ "cacheWrite": 6.25
3912
+ },
3913
+ "contextWindow": 1000000,
3914
+ "maxTokens": 128000,
3915
+ "thinking": {
3916
+ "mode": "anthropic-adaptive",
3917
+ "minLevel": "minimal",
3918
+ "maxLevel": "max"
3919
+ }
3664
3920
  }
3665
3921
  },
3666
3922
  "azure-openai": {
@@ -10255,6 +10511,34 @@
10255
10511
  "minLevel": "minimal",
10256
10512
  "maxLevel": "high"
10257
10513
  }
10514
+ },
10515
+ "claude-opus-5": {
10516
+ "id": "claude-opus-5",
10517
+ "name": "Anthropic Opus 5",
10518
+ "api": "anthropic-messages",
10519
+ "provider": "github-copilot",
10520
+ "baseUrl": "https://api.githubcopilot.com",
10521
+ "reasoning": true,
10522
+ "input": [
10523
+ "text",
10524
+ "image"
10525
+ ],
10526
+ "cost": {
10527
+ "input": 5,
10528
+ "output": 25,
10529
+ "cacheRead": 0.5,
10530
+ "cacheWrite": 6.25
10531
+ },
10532
+ "contextWindow": 1000000,
10533
+ "maxTokens": 64000,
10534
+ "headers": {
10535
+ "User-Agent": "opencode/1.3.15"
10536
+ },
10537
+ "thinking": {
10538
+ "mode": "anthropic-adaptive",
10539
+ "minLevel": "minimal",
10540
+ "maxLevel": "max"
10541
+ }
10258
10542
  }
10259
10543
  },
10260
10544
  "gitlab-duo": {
@@ -22786,6 +23070,46 @@
22786
23070
  "minLevel": "minimal",
22787
23071
  "maxLevel": "xhigh"
22788
23072
  }
23073
+ },
23074
+ "anthropic/claude-opus-5": {
23075
+ "id": "anthropic/claude-opus-5",
23076
+ "name": "Anthropic Opus 5",
23077
+ "api": "openai-completions",
23078
+ "provider": "kilo",
23079
+ "baseUrl": "https://api.kilo.ai/api/gateway",
23080
+ "reasoning": false,
23081
+ "input": [
23082
+ "text",
23083
+ "image"
23084
+ ],
23085
+ "cost": {
23086
+ "input": 0,
23087
+ "output": 0,
23088
+ "cacheRead": 0,
23089
+ "cacheWrite": 0
23090
+ },
23091
+ "contextWindow": 222222,
23092
+ "maxTokens": 8888
23093
+ },
23094
+ "anthropic/claude-opus-5-fast": {
23095
+ "id": "anthropic/claude-opus-5-fast",
23096
+ "name": "Anthropic Opus 5 (Fast) ($$$$)",
23097
+ "api": "openai-completions",
23098
+ "provider": "kilo",
23099
+ "baseUrl": "https://api.kilo.ai/api/gateway",
23100
+ "reasoning": false,
23101
+ "input": [
23102
+ "text",
23103
+ "image"
23104
+ ],
23105
+ "cost": {
23106
+ "input": 0,
23107
+ "output": 0,
23108
+ "cacheRead": 0,
23109
+ "cacheWrite": 0
23110
+ },
23111
+ "contextWindow": 222222,
23112
+ "maxTokens": 8888
22789
23113
  }
22790
23114
  },
22791
23115
  "kimi-code": {
@@ -61331,6 +61655,103 @@
61331
61655
  },
61332
61656
  "contextWindow": 131072,
61333
61657
  "maxTokens": 131072
61658
+ },
61659
+ "claude-opus-5": {
61660
+ "id": "claude-opus-5",
61661
+ "name": "Anthropic Opus 5",
61662
+ "api": "anthropic-messages",
61663
+ "provider": "opencode-zen",
61664
+ "baseUrl": "https://opencode.ai/zen",
61665
+ "reasoning": true,
61666
+ "input": [
61667
+ "text",
61668
+ "image"
61669
+ ],
61670
+ "cost": {
61671
+ "input": 5,
61672
+ "output": 25,
61673
+ "cacheRead": 0.5,
61674
+ "cacheWrite": 6.25
61675
+ },
61676
+ "contextWindow": 1000000,
61677
+ "maxTokens": 128000,
61678
+ "thinking": {
61679
+ "mode": "anthropic-adaptive",
61680
+ "minLevel": "minimal",
61681
+ "maxLevel": "max"
61682
+ }
61683
+ }
61684
+ },
61685
+ "opengateway": {
61686
+ "openai/gpt-4o": {
61687
+ "id": "openai/gpt-4o",
61688
+ "name": "GPT-4o (OpenGateway)",
61689
+ "api": "openai-completions",
61690
+ "provider": "opengateway",
61691
+ "baseUrl": "https://apis.opengateway.ai/v1",
61692
+ "reasoning": false,
61693
+ "input": [
61694
+ "text",
61695
+ "image"
61696
+ ],
61697
+ "cost": {
61698
+ "input": 2.5,
61699
+ "output": 10,
61700
+ "cacheRead": 1.25,
61701
+ "cacheWrite": 0
61702
+ },
61703
+ "contextWindow": 128000,
61704
+ "maxTokens": 16384
61705
+ },
61706
+ "anthropic/claude-sonnet-4-5": {
61707
+ "id": "anthropic/claude-sonnet-4-5",
61708
+ "name": "Anthropic Sonnet 4.5 (OpenGateway)",
61709
+ "api": "openai-completions",
61710
+ "provider": "opengateway",
61711
+ "baseUrl": "https://apis.opengateway.ai/v1",
61712
+ "reasoning": true,
61713
+ "input": [
61714
+ "text",
61715
+ "image"
61716
+ ],
61717
+ "cost": {
61718
+ "input": 3,
61719
+ "output": 15,
61720
+ "cacheRead": 0.3,
61721
+ "cacheWrite": 3.75
61722
+ },
61723
+ "contextWindow": 200000,
61724
+ "maxTokens": 64000,
61725
+ "thinking": {
61726
+ "mode": "effort",
61727
+ "minLevel": "minimal",
61728
+ "maxLevel": "high"
61729
+ }
61730
+ },
61731
+ "google/gemini-2.5-pro": {
61732
+ "id": "google/gemini-2.5-pro",
61733
+ "name": "Gemini 2.5 Pro (OpenGateway)",
61734
+ "api": "openai-completions",
61735
+ "provider": "opengateway",
61736
+ "baseUrl": "https://apis.opengateway.ai/v1",
61737
+ "reasoning": true,
61738
+ "input": [
61739
+ "text",
61740
+ "image"
61741
+ ],
61742
+ "cost": {
61743
+ "input": 1.25,
61744
+ "output": 10,
61745
+ "cacheRead": 0.31,
61746
+ "cacheWrite": 0
61747
+ },
61748
+ "contextWindow": 1048576,
61749
+ "maxTokens": 65536,
61750
+ "thinking": {
61751
+ "mode": "effort",
61752
+ "minLevel": "minimal",
61753
+ "maxLevel": "high"
61754
+ }
61334
61755
  }
61335
61756
  },
61336
61757
  "openrouter": {
@@ -69920,6 +70341,56 @@
69920
70341
  "minLevel": "minimal",
69921
70342
  "maxLevel": "high"
69922
70343
  }
70344
+ },
70345
+ "anthropic/claude-opus-5": {
70346
+ "id": "anthropic/claude-opus-5",
70347
+ "name": "Anthropic Opus 5",
70348
+ "api": "openai-completions",
70349
+ "provider": "openrouter",
70350
+ "baseUrl": "https://openrouter.ai/api/v1",
70351
+ "reasoning": true,
70352
+ "input": [
70353
+ "text",
70354
+ "image"
70355
+ ],
70356
+ "cost": {
70357
+ "input": 5,
70358
+ "output": 25,
70359
+ "cacheRead": 0.5,
70360
+ "cacheWrite": 6.25
70361
+ },
70362
+ "contextWindow": 1000000,
70363
+ "maxTokens": 128000,
70364
+ "thinking": {
70365
+ "mode": "effort",
70366
+ "minLevel": "minimal",
70367
+ "maxLevel": "high"
70368
+ }
70369
+ },
70370
+ "anthropic/claude-opus-5-fast": {
70371
+ "id": "anthropic/claude-opus-5-fast",
70372
+ "name": "Anthropic Opus 5 (Fast)",
70373
+ "api": "openai-completions",
70374
+ "provider": "openrouter",
70375
+ "baseUrl": "https://openrouter.ai/api/v1",
70376
+ "reasoning": true,
70377
+ "input": [
70378
+ "text",
70379
+ "image"
70380
+ ],
70381
+ "cost": {
70382
+ "input": 10,
70383
+ "output": 50,
70384
+ "cacheRead": 1,
70385
+ "cacheWrite": 12.5
70386
+ },
70387
+ "contextWindow": 1000000,
70388
+ "maxTokens": 128000,
70389
+ "thinking": {
70390
+ "mode": "effort",
70391
+ "minLevel": "minimal",
70392
+ "maxLevel": "high"
70393
+ }
69923
70394
  }
69924
70395
  },
69925
70396
  "qianfan": {
@@ -73537,6 +74008,57 @@
73537
74008
  "compat": {
73538
74009
  "supportsUsageInStreaming": false
73539
74010
  }
74011
+ },
74012
+ "claude-opus-5": {
74013
+ "id": "claude-opus-5",
74014
+ "name": "Anthropic Opus 5",
74015
+ "api": "openai-completions",
74016
+ "provider": "venice",
74017
+ "baseUrl": "https://api.venice.ai/api/v1",
74018
+ "reasoning": true,
74019
+ "input": [
74020
+ "text",
74021
+ "image"
74022
+ ],
74023
+ "cost": {
74024
+ "input": 0,
74025
+ "output": 0,
74026
+ "cacheRead": 0,
74027
+ "cacheWrite": 0
74028
+ },
74029
+ "contextWindow": 1000000,
74030
+ "maxTokens": 128000,
74031
+ "compat": {
74032
+ "supportsUsageInStreaming": false
74033
+ },
74034
+ "thinking": {
74035
+ "mode": "effort",
74036
+ "minLevel": "minimal",
74037
+ "maxLevel": "xhigh"
74038
+ }
74039
+ },
74040
+ "claude-opus-5-fast": {
74041
+ "id": "claude-opus-5-fast",
74042
+ "name": "anthropic-opus-5-fast",
74043
+ "api": "openai-completions",
74044
+ "provider": "venice",
74045
+ "baseUrl": "https://api.venice.ai/api/v1",
74046
+ "reasoning": false,
74047
+ "input": [
74048
+ "text",
74049
+ "image"
74050
+ ],
74051
+ "cost": {
74052
+ "input": 0,
74053
+ "output": 0,
74054
+ "cacheRead": 0,
74055
+ "cacheWrite": 0
74056
+ },
74057
+ "contextWindow": 1000000,
74058
+ "maxTokens": 8888,
74059
+ "compat": {
74060
+ "supportsUsageInStreaming": false
74061
+ }
73540
74062
  }
73541
74063
  },
73542
74064
  "vercel-ai-gateway": {
@@ -78781,6 +79303,56 @@
78781
79303
  "minLevel": "minimal",
78782
79304
  "maxLevel": "xhigh"
78783
79305
  }
79306
+ },
79307
+ "anthropic/claude-opus-5": {
79308
+ "id": "anthropic/claude-opus-5",
79309
+ "name": "Anthropic Opus 5",
79310
+ "api": "anthropic-messages",
79311
+ "provider": "vercel-ai-gateway",
79312
+ "baseUrl": "https://ai-gateway.vercel.sh",
79313
+ "reasoning": true,
79314
+ "input": [
79315
+ "text",
79316
+ "image"
79317
+ ],
79318
+ "cost": {
79319
+ "input": 5,
79320
+ "output": 25,
79321
+ "cacheRead": 0.5,
79322
+ "cacheWrite": 6.25
79323
+ },
79324
+ "contextWindow": 1000000,
79325
+ "maxTokens": 128000,
79326
+ "thinking": {
79327
+ "mode": "anthropic-adaptive",
79328
+ "minLevel": "minimal",
79329
+ "maxLevel": "max"
79330
+ }
79331
+ },
79332
+ "anthropic/claude-opus-5-fast": {
79333
+ "id": "anthropic/claude-opus-5-fast",
79334
+ "name": "Anthropic Opus 5 (Fast)",
79335
+ "api": "anthropic-messages",
79336
+ "provider": "vercel-ai-gateway",
79337
+ "baseUrl": "https://ai-gateway.vercel.sh",
79338
+ "reasoning": true,
79339
+ "input": [
79340
+ "text",
79341
+ "image"
79342
+ ],
79343
+ "cost": {
79344
+ "input": 10,
79345
+ "output": 50,
79346
+ "cacheRead": 1,
79347
+ "cacheWrite": 12.5
79348
+ },
79349
+ "contextWindow": 1000000,
79350
+ "maxTokens": 128000,
79351
+ "thinking": {
79352
+ "mode": "anthropic-adaptive",
79353
+ "minLevel": "minimal",
79354
+ "maxLevel": "max"
79355
+ }
78784
79356
  }
78785
79357
  },
78786
79358
  "xai": {
@@ -84779,4 +85351,4 @@
84779
85351
  }
84780
85352
  }
84781
85353
  }
84782
- }
85354
+ }
@@ -33,6 +33,7 @@ import {
33
33
  openaiModelManagerOptions,
34
34
  opencodeGoModelManagerOptions,
35
35
  opencodeZenModelManagerOptions,
36
+ opengatewayModelManagerOptions,
36
37
  openrouterModelManagerOptions,
37
38
  qianfanModelManagerOptions,
38
39
  qwenPortalModelManagerOptions,
@@ -312,6 +313,12 @@ export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [
312
313
  config => zenmuxModelManagerOptions(config),
313
314
  catalog("ZenMux", ["ZENMUX_API_KEY"]),
314
315
  ),
316
+ catalogDescriptor(
317
+ "opengateway",
318
+ "openai/gpt-4o",
319
+ config => opengatewayModelManagerOptions(config),
320
+ catalog("OpenGateway by Sionic AI", ["OPENGATEWAY_API_KEY"]),
321
+ ),
315
322
  catalogDescriptor("zai", "glm-5.2", config => zaiModelManagerOptions(config), catalog("zAI", ["ZAI_API_KEY"])),
316
323
  catalogDescriptor(
317
324
  "glm-zcode",
@@ -1099,6 +1099,26 @@ export function zenmuxModelManagerOptions(config?: ZenMuxModelManagerConfig): Mo
1099
1099
  };
1100
1100
  }
1101
1101
 
1102
+ // ---------------------------------------------------------------------------
1103
+ // 10.5.1 OpenGateway by Sionic AI
1104
+ // ---------------------------------------------------------------------------
1105
+
1106
+ export interface OpenGatewayModelManagerConfig {
1107
+ apiKey?: string;
1108
+ baseUrl?: string;
1109
+ }
1110
+
1111
+ /**
1112
+ * OpenGateway by Sionic AI — an OpenAI-compatible gateway that fronts OpenAI,
1113
+ * Anthropic, and Google models behind one API key. Models are discovered from
1114
+ * the OpenAI-compatible `/v1/models` endpoint.
1115
+ */
1116
+ export function opengatewayModelManagerOptions(
1117
+ config?: OpenGatewayModelManagerConfig,
1118
+ ): ModelManagerOptions<"openai-completions"> {
1119
+ return createSimpleOpenAICompletionsOptions("opengateway", "https://apis.opengateway.ai/v1", config);
1120
+ }
1121
+
1102
1122
  // ---------------------------------------------------------------------------
1103
1123
  // 10.6 Kilo Gateway
1104
1124
  // ---------------------------------------------------------------------------
@@ -60,7 +60,12 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
60
60
  import { transportFailureFacts } from "../utils/fallback-transport";
61
61
  import { isFoundryEnabled } from "../utils/foundry";
62
62
  import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } from "../utils/http-inspector";
63
- import { getStreamFirstEventTimeoutMs, getStreamIdleTimeoutMs, iterateWithIdleTimeout } from "../utils/idle-iterator";
63
+ import {
64
+ getProviderFirstEventTimeoutFallbackMs,
65
+ getStreamFirstEventTimeoutMs,
66
+ getStreamIdleTimeoutMs,
67
+ iterateWithIdleTimeout,
68
+ } from "../utils/idle-iterator";
64
69
  import { parseJsonWithRepair, parseStreamingJson } from "../utils/json-parse";
65
70
  import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
66
71
  import { notifyProviderResponse } from "../utils/provider-response";
@@ -1416,7 +1421,9 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1416
1421
  firstTokenTime = undefined;
1417
1422
  };
1418
1423
  const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs();
1419
- const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs);
1424
+ const firstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
1425
+ const firstEventTimeoutMs =
1426
+ options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs);
1420
1427
  stream.push({ type: "start", partial: output });
1421
1428
  // Retry loop for transient errors from the stream.
1422
1429
  // Provider-level transport/rate-limit failures: only before any streamed content starts.
@@ -89,6 +89,8 @@ export function streamOpenAIAnthropicShim(
89
89
  onResponse: options?.onResponse,
90
90
  onSseEvent: options?.onSseEvent,
91
91
  fetch: options?.fetch,
92
+ streamIdleTimeoutMs: options?.streamIdleTimeoutMs,
93
+ streamFirstEventTimeoutMs: options?.streamFirstEventTimeoutMs,
92
94
  thinkingEnabled,
93
95
  thinkingBudgetTokens: thinkingBudget,
94
96
  });
@@ -118,6 +120,8 @@ export function streamOpenAIAnthropicShim(
118
120
  onResponse: options?.onResponse,
119
121
  onSseEvent: options?.onSseEvent,
120
122
  fetch: options?.fetch,
123
+ streamIdleTimeoutMs: options?.streamIdleTimeoutMs,
124
+ streamFirstEventTimeoutMs: options?.streamFirstEventTimeoutMs,
121
125
  reasoning: reasoningEffort,
122
126
  });
123
127
 
@@ -49,6 +49,7 @@ import {
49
49
  import {
50
50
  createWatchdog,
51
51
  getOpenAIStreamIdleTimeoutMs,
52
+ getProviderFirstEventTimeoutFallbackMs,
52
53
  getStreamFirstEventTimeoutMs,
53
54
  iterateWithIdleTimeout,
54
55
  } from "../utils/idle-iterator";
@@ -427,6 +428,7 @@ function getTrailingPartialDeepseekToken(text: string): string {
427
428
  }
428
429
 
429
430
  const ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS = 300_000;
431
+
430
432
  const OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE =
431
433
  "OpenAI completions stream timed out while waiting for the first event";
432
434
 
@@ -564,7 +566,9 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
564
566
  }
565
567
  }
566
568
  const firstEventFallbackMs =
567
- model.provider === "alibaba-token-plan" ? ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS : undefined;
569
+ model.provider === "alibaba-token-plan"
570
+ ? ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS
571
+ : getProviderFirstEventTimeoutFallbackMs(model.provider);
568
572
  const firstEventWatchdog = createWatchdog(
569
573
  options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
570
574
  () => abortTracker.abortLocally(firstEventTimeoutAbortError),
@@ -190,6 +190,22 @@ interface LazyStreamLimits {
190
190
  const GOOGLE_GEMINI_CLI_LAZY_STREAM_LIMITS: LazyStreamLimits = {
191
191
  defaultFirstEventTimeoutMs: 300_000,
192
192
  };
193
+ const SLOW_FIRST_EVENT_PROVIDERS = new Set(["alibaba-token-plan", "kimi-code"]);
194
+
195
+ /**
196
+ * Resolves the first-event timeout fallback for the outer lazy-stream watchdog.
197
+ * A configured wrapper-specific fallback (from `LazyStreamLimits`) always wins;
198
+ * otherwise providers known to have slow first events get a five-minute floor
199
+ * matching their inner provider-level override. Returns `undefined` for
200
+ * providers that should use the shared default.
201
+ */
202
+ export function resolveLazyStreamFirstEventFallbackMs(
203
+ provider: string,
204
+ configuredFallbackMs?: number,
205
+ ): number | undefined {
206
+ if (configuredFallbackMs !== undefined) return configuredFallbackMs;
207
+ return SLOW_FIRST_EVENT_PROVIDERS.has(provider) ? 300_000 : undefined;
208
+ }
193
209
 
194
210
  function forwardStream<TApi extends Api>(
195
211
  target: EventStreamImpl,
@@ -202,11 +218,14 @@ function forwardStream<TApi extends Api>(
202
218
  (async () => {
203
219
  try {
204
220
  const idleTimeoutMs = options.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs(limits?.defaultIdleTimeoutMs);
221
+ const firstEventFallbackMs = resolveLazyStreamFirstEventFallbackMs(
222
+ model.provider,
223
+ limits?.defaultFirstEventTimeoutMs,
224
+ );
205
225
  const watchedSource = iterateWithIdleTimeout(source, {
206
226
  idleTimeoutMs,
207
227
  firstItemTimeoutMs:
208
- options.streamFirstEventTimeoutMs ??
209
- getStreamFirstEventTimeoutMs(idleTimeoutMs, limits?.defaultFirstEventTimeoutMs),
228
+ options.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
210
229
  errorMessage: LAZY_STREAM_IDLE_TIMEOUT_ERROR,
211
230
  firstItemErrorMessage: LAZY_STREAM_FIRST_EVENT_TIMEOUT_ERROR,
212
231
  onIdle: () => abortTracker.abortLocally(new Error(LAZY_STREAM_IDLE_TIMEOUT_ERROR)),
package/src/stream.ts CHANGED
@@ -162,6 +162,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
162
162
  "qwen-portal": () => $pickCredentialEnv("QWEN_OAUTH_TOKEN", "QWEN_PORTAL_API_KEY"),
163
163
  together: "TOGETHER_API_KEY",
164
164
  zenmux: "ZENMUX_API_KEY",
165
+ opengateway: "OPENGATEWAY_API_KEY",
165
166
  venice: "VENICE_API_KEY",
166
167
  vllm: "VLLM_API_KEY",
167
168
  xiaomi: "XIAOMI_API_KEY",
package/src/types.ts CHANGED
@@ -147,6 +147,7 @@ export type KnownProvider =
147
147
  | "minimax"
148
148
  | "opencode-go"
149
149
  | "opencode-zen"
150
+ | "opengateway"
150
151
  | "synthetic"
151
152
  | "cloudflare-ai-gateway"
152
153
  | "huggingface"
@@ -2,6 +2,11 @@ import { $env } from "@gajae-code/utils";
2
2
 
3
3
  const DEFAULT_STREAM_IDLE_TIMEOUT_MS = 120_000;
4
4
  const DEFAULT_STREAM_FIRST_EVENT_TIMEOUT_MS = 100_000;
5
+ const KIMI_CODE_FIRST_EVENT_TIMEOUT_MS = 300_000;
6
+
7
+ export function getProviderFirstEventTimeoutFallbackMs(provider: string): number | undefined {
8
+ return provider === "kimi-code" ? KIMI_CODE_FIRST_EVENT_TIMEOUT_MS : undefined;
9
+ }
5
10
 
6
11
  function normalizeIdleTimeoutMs(value: string | undefined, fallback: number): number | undefined {
7
12
  if (value === undefined) return fallback;
@@ -240,6 +240,11 @@ const builtInOAuthProviders: OAuthProviderInfo[] = [
240
240
  name: "ZenMux",
241
241
  available: true,
242
242
  },
243
+ {
244
+ id: "opengateway",
245
+ name: "OpenGateway by Sionic AI",
246
+ available: true,
247
+ },
243
248
  {
244
249
  id: "vllm",
245
250
  name: "vLLM (Local OpenAI-compatible)",
@@ -386,6 +391,7 @@ export async function refreshOAuthToken(
386
391
  case "vercel-ai-gateway":
387
392
  case "qwen-portal":
388
393
  case "zenmux":
394
+ case "opengateway":
389
395
  case "vllm":
390
396
  // API keys / static bearer tokens don't expire, return as-is
391
397
  newCredentials = credentials;
@@ -0,0 +1,15 @@
1
+ /** OpenGateway (by Sionic AI) login flow (API key paste, validated via /v1/models). */
2
+ import { createApiKeyLogin } from "./api-key-login";
3
+
4
+ export const loginOpenGateway = createApiKeyLogin({
5
+ providerLabel: "OpenGateway by Sionic AI",
6
+ authUrl: "https://opengateway.ai/dashboard",
7
+ instructions: "Create or copy your OpenGateway API key",
8
+ promptMessage: "Paste your OpenGateway API key",
9
+ placeholder: "sk-...",
10
+ validation: {
11
+ kind: "models-endpoint",
12
+ provider: "OpenGateway by Sionic AI",
13
+ modelsUrl: "https://apis.opengateway.ai/v1/models",
14
+ },
15
+ });
@@ -40,6 +40,7 @@ export type OAuthProvider =
40
40
  | "openai-codex-device"
41
41
  | "opencode-go"
42
42
  | "opencode-zen"
43
+ | "opengateway"
43
44
  | "parallel"
44
45
  | "perplexity"
45
46
  | "qianfan"