@gajae-code/ai 0.11.8 → 0.11.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,12 +2,21 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.11.9] - 2026-07-24
6
+ ### Fixed
7
+
8
+ - Kimi Code now allows one continuous 300-second first-event wait before aborting, while preserving explicit caller and environment timeout overrides and the existing inter-event idle timeout.
9
+
10
+ ### Added
11
+
12
+ - Added first-class support for **OpenGateway by Sionic AI**, an OpenAI-compatible gateway. Registers the `opengateway` provider descriptor, `/login` OAuth entry (API-key paste validated against `https://apis.opengateway.ai/v1/models`), `OPENGATEWAY_API_KEY` environment resolution, and bundled `models.json` seed models. Models are discovered dynamically from the OpenAI-compatible `/v1/models` endpoint (base URL `https://apis.opengateway.ai/v1`).
13
+
5
14
  ## [0.11.8] - 2026-07-23
6
15
 
7
16
  ### Fixed
8
17
 
9
18
  - OpenAI Responses / Codex native history replay no longer submits missing resident-image placeholders as `input_image.image_url`. Invalid values (including `[Session resident imageUrl blob missing: …]`) are dropped, or retained as `file_id`-only parts when a non-empty `file_id` is present, so a single unavailable historical image cannot brick `/retry` (#2924).
10
- - Raised the first-event stream timeout floor to five minutes for `alibaba-token-plan` models using the OpenAI Completions and Responses APIs, while preserving caller and environment overrides and the existing inter-event idle timeout.
19
+ - Raised the first-event stream timeout floor to five minutes for `alibaba-token-plan` models at both the OpenAI provider and outer lazy-stream watchdogs, while preserving caller and environment overrides and the existing inter-event idle timeout.
11
20
  - OAuth refresh peer-rotation recovery now runs before failure classification instead of only on the definitive-failure path, and the definitive matcher recognizes the "grant is invalid" phrasing. Providers whose invalid-grant response does not contain the literal `invalid_grant` (e.g. Kimi's 400 "The provided authorization grant is invalid") previously had rotation races misclassified as transient, temp-blocking a healthy credential for five minutes on every race; with Kimi's ~12-minute access tokens and multiple processes sharing the credential store this surfaced as repeated logouts. Genuine revocations are now disabled with a cause instead of looping temp-blocks.
12
21
  - Anthropic 400 `Invalid \`signature\` in \`thinking\` block` responses now trigger the one-shot thinking replay repair instead of failing the turn. The existing repair matcher only recognized the "latest assistant message ... cannot be modified" wording, so the signature-validation variant — which can cite a `thinking`/`redacted_thinking` block anywhere in the replayed history (e.g. after compaction/pruning rewrote an earlier turn) — was treated as a fatal request error. The retry now rebuilds the request with thinking blocks dropped from every replayed assistant message (`repairAllAssistantThinking`), while the latest-message mutation variant keeps the targeted latest-only repair.
13
22
 
package/README.md CHANGED
@@ -69,6 +69,7 @@ Unified LLM API with automatic model discovery, provider configuration, token an
69
69
  - **MiniMax Coding Plan** (requires `MINIMAX_CODE_API_KEY` or `MINIMAX_CODE_CN_API_KEY`)
70
70
  - **Xiaomi MiMo** (requires `XIAOMI_API_KEY`)
71
71
  - **ZenMux** (requires `ZENMUX_API_KEY`)
72
+ - **OpenGateway by Sionic AI** (requires `OPENGATEWAY_API_KEY`)
72
73
  - **Qwen Portal** (supports `QWEN_OAUTH_TOKEN` or `QWEN_PORTAL_API_KEY`)
73
74
  - **Cloudflare AI Gateway** (requires `CLOUDFLARE_AI_GATEWAY_API_KEY` and provider-specific gateway base URL)
74
75
  - **Ollama** (local OpenAI-compatible runtime; optional `OLLAMA_API_KEY`)
@@ -954,6 +955,7 @@ In Node.js environments, you can set environment variables to avoid passing API
954
955
  | MiniMax Code | `MINIMAX_CODE_API_KEY` (international) or `MINIMAX_CODE_CN_API_KEY` (China) |
955
956
  | Xiaomi MiMo | `XIAOMI_API_KEY` |
956
957
  | ZenMux | `ZENMUX_API_KEY` |
958
+ | OpenGateway | `OPENGATEWAY_API_KEY` |
957
959
  | vLLM | `VLLM_API_KEY` |
958
960
  | Cloudflare AI Gateway | `CLOUDFLARE_AI_GATEWAY_API_KEY` |
959
961
  | GitHub Copilot | `COPILOT_GITHUB_TOKEN` or `GH_TOKEN` or `GITHUB_TOKEN` |
@@ -977,6 +979,7 @@ Provider endpoint defaults for the current OpenAI-compatible integrations:
977
979
  - Xiaomi MiMo: `https://api.xiaomimimo.com/anthropic`
978
980
  - ZenMux (OpenAI): `https://zenmux.ai/api/v1`
979
981
  - ZenMux (Anthropic models): `https://zenmux.ai/api/anthropic`
982
+ - OpenGateway by Sionic AI: `https://apis.opengateway.ai/v1`
980
983
  - vLLM: `http://127.0.0.1:8000/v1`
981
984
  - Ollama: local OpenAI-compatible runtime (`http://127.0.0.1:11434/v1`)
982
985
  - Ollama Cloud: native Ollama API host (`https://ollama.com/api`, configured here as base URL `https://ollama.com`)
@@ -111,6 +111,16 @@ export interface ZenMuxModelManagerConfig {
111
111
  baseUrl?: string;
112
112
  }
113
113
  export declare function zenmuxModelManagerOptions(config?: ZenMuxModelManagerConfig): ModelManagerOptions<Api>;
114
+ export interface OpenGatewayModelManagerConfig {
115
+ apiKey?: string;
116
+ baseUrl?: string;
117
+ }
118
+ /**
119
+ * OpenGateway by Sionic AI — an OpenAI-compatible gateway that fronts OpenAI,
120
+ * Anthropic, and Google models behind one API key. Models are discovered from
121
+ * the OpenAI-compatible `/v1/models` endpoint.
122
+ */
123
+ export declare function opengatewayModelManagerOptions(config?: OpenGatewayModelManagerConfig): ModelManagerOptions<"openai-completions">;
114
124
  export interface KiloModelManagerConfig {
115
125
  apiKey?: string;
116
126
  baseUrl?: string;
@@ -17,6 +17,14 @@ interface BedrockProviderModule {
17
17
  streamBedrock: (model: Model<"bedrock-converse-stream">, context: Context, options: BedrockOptions) => AssistantMessageEventStream;
18
18
  }
19
19
  export declare function setBedrockProviderModule(module: BedrockProviderModule): void;
20
+ /**
21
+ * Resolves the first-event timeout fallback for the outer lazy-stream watchdog.
22
+ * A configured wrapper-specific fallback (from `LazyStreamLimits`) always wins;
23
+ * otherwise providers known to have slow first events get a five-minute floor
24
+ * matching their inner provider-level override. Returns `undefined` for
25
+ * providers that should use the shared default.
26
+ */
27
+ export declare function resolveLazyStreamFirstEventFallbackMs(provider: string, configuredFallbackMs?: number): number | undefined;
20
28
  export declare const streamAnthropic: (model: Model<"anthropic-messages">, context: Context, options: OptionsForApi<"anthropic-messages">) => EventStreamImpl;
21
29
  export declare const streamAzureOpenAIResponses: (model: Model<"azure-openai-responses">, context: Context, options: OptionsForApi<"azure-openai-responses">) => EventStreamImpl;
22
30
  export declare const streamGoogle: (model: Model<"google-generative-ai">, context: Context, options: OptionsForApi<"google-generative-ai">) => EventStreamImpl;
@@ -51,7 +51,7 @@ export interface ThinkingConfig {
51
51
  /** Provider-specific transport used to encode the selected effort. */
52
52
  mode: ThinkingControlMode;
53
53
  }
54
- export type KnownProvider = "alibaba-token-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
54
+ export type KnownProvider = "alibaba-token-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "opengateway" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
55
55
  export type Provider = KnownProvider | string;
56
56
  import type { Effort } from "./model-thinking";
57
57
  /** Token budgets for each thinking level (token-based providers only) */
@@ -1,3 +1,4 @@
1
+ export declare function getProviderFirstEventTimeoutFallbackMs(provider: string): number | undefined;
1
2
  /**
2
3
  * Returns the idle timeout used for provider streaming transports.
3
4
  *
@@ -0,0 +1 @@
1
+ export declare const loginOpenGateway: (options: import("./types").OAuthController) => Promise<string>;
@@ -7,7 +7,7 @@ export type OAuthCredentials = {
7
7
  email?: string;
8
8
  accountId?: string;
9
9
  };
10
- export type OAuthProvider = "alibaba-token-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
10
+ export type OAuthProvider = "alibaba-token-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "opengateway" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
11
11
  export type OAuthProviderId = OAuthProvider | (string & {});
12
12
  export type OAuthPrompt = {
13
13
  message: string;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/ai",
4
- "version": "0.11.8",
4
+ "version": "0.11.9",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://gajae-code.com",
7
7
  "author": "Yeachan-Heo and Gajae Code Contributors",
@@ -40,7 +40,7 @@
40
40
  "dependencies": {
41
41
  "@anthropic-ai/sdk": "^0.94.0",
42
42
  "@bufbuild/protobuf": "^2.12.0",
43
- "@gajae-code/utils": "0.11.8",
43
+ "@gajae-code/utils": "0.11.9",
44
44
  "openai": "^6.36.0",
45
45
  "partial-json": "^0.1.7",
46
46
  "zod": "4.4.3"
@@ -1998,6 +1998,12 @@ export class AuthStorage {
1998
1998
  await saveApiKeyCredential(apiKey);
1999
1999
  return;
2000
2000
  }
2001
+ case "opengateway": {
2002
+ const { loginOpenGateway } = await import("./utils/oauth/opengateway");
2003
+ const apiKey = await loginOpenGateway(ctrl);
2004
+ await saveApiKeyCredential(apiKey);
2005
+ return;
2006
+ }
2001
2007
  default: {
2002
2008
  const customProvider = getOAuthProvider(provider);
2003
2009
  if (!customProvider) {
package/src/cli.ts CHANGED
@@ -118,6 +118,7 @@ Providers:
118
118
  minimax-code-cn MiniMax Coding Plan (China)
119
119
  cursor Cursor (Anthropic, GPT, etc.)
120
120
  zenmux ZenMux
121
+ opengateway OpenGateway by Sionic AI
121
122
  ollama-cloud Ollama Cloud
122
123
 
123
124
  Examples:
package/src/models.json CHANGED
@@ -61333,6 +61333,78 @@
61333
61333
  "maxTokens": 131072
61334
61334
  }
61335
61335
  },
61336
+ "opengateway": {
61337
+ "openai/gpt-4o": {
61338
+ "id": "openai/gpt-4o",
61339
+ "name": "GPT-4o (OpenGateway)",
61340
+ "api": "openai-completions",
61341
+ "provider": "opengateway",
61342
+ "baseUrl": "https://apis.opengateway.ai/v1",
61343
+ "reasoning": false,
61344
+ "input": [
61345
+ "text",
61346
+ "image"
61347
+ ],
61348
+ "cost": {
61349
+ "input": 2.5,
61350
+ "output": 10,
61351
+ "cacheRead": 1.25,
61352
+ "cacheWrite": 0
61353
+ },
61354
+ "contextWindow": 128000,
61355
+ "maxTokens": 16384
61356
+ },
61357
+ "anthropic/claude-sonnet-4-5": {
61358
+ "id": "anthropic/claude-sonnet-4-5",
61359
+ "name": "Anthropic Sonnet 4.5 (OpenGateway)",
61360
+ "api": "openai-completions",
61361
+ "provider": "opengateway",
61362
+ "baseUrl": "https://apis.opengateway.ai/v1",
61363
+ "reasoning": true,
61364
+ "input": [
61365
+ "text",
61366
+ "image"
61367
+ ],
61368
+ "cost": {
61369
+ "input": 3,
61370
+ "output": 15,
61371
+ "cacheRead": 0.3,
61372
+ "cacheWrite": 3.75
61373
+ },
61374
+ "contextWindow": 200000,
61375
+ "maxTokens": 64000,
61376
+ "thinking": {
61377
+ "mode": "effort",
61378
+ "minLevel": "minimal",
61379
+ "maxLevel": "high"
61380
+ }
61381
+ },
61382
+ "google/gemini-2.5-pro": {
61383
+ "id": "google/gemini-2.5-pro",
61384
+ "name": "Gemini 2.5 Pro (OpenGateway)",
61385
+ "api": "openai-completions",
61386
+ "provider": "opengateway",
61387
+ "baseUrl": "https://apis.opengateway.ai/v1",
61388
+ "reasoning": true,
61389
+ "input": [
61390
+ "text",
61391
+ "image"
61392
+ ],
61393
+ "cost": {
61394
+ "input": 1.25,
61395
+ "output": 10,
61396
+ "cacheRead": 0.31,
61397
+ "cacheWrite": 0
61398
+ },
61399
+ "contextWindow": 1048576,
61400
+ "maxTokens": 65536,
61401
+ "thinking": {
61402
+ "mode": "effort",
61403
+ "minLevel": "minimal",
61404
+ "maxLevel": "high"
61405
+ }
61406
+ }
61407
+ },
61336
61408
  "openrouter": {
61337
61409
  "~anthropic/claude-fable-latest": {
61338
61410
  "id": "~anthropic/claude-fable-latest",
@@ -33,6 +33,7 @@ import {
33
33
  openaiModelManagerOptions,
34
34
  opencodeGoModelManagerOptions,
35
35
  opencodeZenModelManagerOptions,
36
+ opengatewayModelManagerOptions,
36
37
  openrouterModelManagerOptions,
37
38
  qianfanModelManagerOptions,
38
39
  qwenPortalModelManagerOptions,
@@ -312,6 +313,12 @@ export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [
312
313
  config => zenmuxModelManagerOptions(config),
313
314
  catalog("ZenMux", ["ZENMUX_API_KEY"]),
314
315
  ),
316
+ catalogDescriptor(
317
+ "opengateway",
318
+ "openai/gpt-4o",
319
+ config => opengatewayModelManagerOptions(config),
320
+ catalog("OpenGateway by Sionic AI", ["OPENGATEWAY_API_KEY"]),
321
+ ),
315
322
  catalogDescriptor("zai", "glm-5.2", config => zaiModelManagerOptions(config), catalog("zAI", ["ZAI_API_KEY"])),
316
323
  catalogDescriptor(
317
324
  "glm-zcode",
@@ -1099,6 +1099,26 @@ export function zenmuxModelManagerOptions(config?: ZenMuxModelManagerConfig): Mo
1099
1099
  };
1100
1100
  }
1101
1101
 
1102
+ // ---------------------------------------------------------------------------
1103
+ // 10.5.1 OpenGateway by Sionic AI
1104
+ // ---------------------------------------------------------------------------
1105
+
1106
+ export interface OpenGatewayModelManagerConfig {
1107
+ apiKey?: string;
1108
+ baseUrl?: string;
1109
+ }
1110
+
1111
+ /**
1112
+ * OpenGateway by Sionic AI — an OpenAI-compatible gateway that fronts OpenAI,
1113
+ * Anthropic, and Google models behind one API key. Models are discovered from
1114
+ * the OpenAI-compatible `/v1/models` endpoint.
1115
+ */
1116
+ export function opengatewayModelManagerOptions(
1117
+ config?: OpenGatewayModelManagerConfig,
1118
+ ): ModelManagerOptions<"openai-completions"> {
1119
+ return createSimpleOpenAICompletionsOptions("opengateway", "https://apis.opengateway.ai/v1", config);
1120
+ }
1121
+
1102
1122
  // ---------------------------------------------------------------------------
1103
1123
  // 10.6 Kilo Gateway
1104
1124
  // ---------------------------------------------------------------------------
@@ -60,7 +60,12 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
60
60
  import { transportFailureFacts } from "../utils/fallback-transport";
61
61
  import { isFoundryEnabled } from "../utils/foundry";
62
62
  import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } from "../utils/http-inspector";
63
- import { getStreamFirstEventTimeoutMs, getStreamIdleTimeoutMs, iterateWithIdleTimeout } from "../utils/idle-iterator";
63
+ import {
64
+ getProviderFirstEventTimeoutFallbackMs,
65
+ getStreamFirstEventTimeoutMs,
66
+ getStreamIdleTimeoutMs,
67
+ iterateWithIdleTimeout,
68
+ } from "../utils/idle-iterator";
64
69
  import { parseJsonWithRepair, parseStreamingJson } from "../utils/json-parse";
65
70
  import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
66
71
  import { notifyProviderResponse } from "../utils/provider-response";
@@ -1416,7 +1421,9 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1416
1421
  firstTokenTime = undefined;
1417
1422
  };
1418
1423
  const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs();
1419
- const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs);
1424
+ const firstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
1425
+ const firstEventTimeoutMs =
1426
+ options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs);
1420
1427
  stream.push({ type: "start", partial: output });
1421
1428
  // Retry loop for transient errors from the stream.
1422
1429
  // Provider-level transport/rate-limit failures: only before any streamed content starts.
@@ -89,6 +89,8 @@ export function streamOpenAIAnthropicShim(
89
89
  onResponse: options?.onResponse,
90
90
  onSseEvent: options?.onSseEvent,
91
91
  fetch: options?.fetch,
92
+ streamIdleTimeoutMs: options?.streamIdleTimeoutMs,
93
+ streamFirstEventTimeoutMs: options?.streamFirstEventTimeoutMs,
92
94
  thinkingEnabled,
93
95
  thinkingBudgetTokens: thinkingBudget,
94
96
  });
@@ -118,6 +120,8 @@ export function streamOpenAIAnthropicShim(
118
120
  onResponse: options?.onResponse,
119
121
  onSseEvent: options?.onSseEvent,
120
122
  fetch: options?.fetch,
123
+ streamIdleTimeoutMs: options?.streamIdleTimeoutMs,
124
+ streamFirstEventTimeoutMs: options?.streamFirstEventTimeoutMs,
121
125
  reasoning: reasoningEffort,
122
126
  });
123
127
 
@@ -49,6 +49,7 @@ import {
49
49
  import {
50
50
  createWatchdog,
51
51
  getOpenAIStreamIdleTimeoutMs,
52
+ getProviderFirstEventTimeoutFallbackMs,
52
53
  getStreamFirstEventTimeoutMs,
53
54
  iterateWithIdleTimeout,
54
55
  } from "../utils/idle-iterator";
@@ -427,6 +428,7 @@ function getTrailingPartialDeepseekToken(text: string): string {
427
428
  }
428
429
 
429
430
  const ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS = 300_000;
431
+
430
432
  const OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE =
431
433
  "OpenAI completions stream timed out while waiting for the first event";
432
434
 
@@ -564,7 +566,9 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
564
566
  }
565
567
  }
566
568
  const firstEventFallbackMs =
567
- model.provider === "alibaba-token-plan" ? ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS : undefined;
569
+ model.provider === "alibaba-token-plan"
570
+ ? ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS
571
+ : getProviderFirstEventTimeoutFallbackMs(model.provider);
568
572
  const firstEventWatchdog = createWatchdog(
569
573
  options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
570
574
  () => abortTracker.abortLocally(firstEventTimeoutAbortError),
@@ -190,6 +190,22 @@ interface LazyStreamLimits {
190
190
  const GOOGLE_GEMINI_CLI_LAZY_STREAM_LIMITS: LazyStreamLimits = {
191
191
  defaultFirstEventTimeoutMs: 300_000,
192
192
  };
193
+ const SLOW_FIRST_EVENT_PROVIDERS = new Set(["alibaba-token-plan", "kimi-code"]);
194
+
195
+ /**
196
+ * Resolves the first-event timeout fallback for the outer lazy-stream watchdog.
197
+ * A configured wrapper-specific fallback (from `LazyStreamLimits`) always wins;
198
+ * otherwise providers known to have slow first events get a five-minute floor
199
+ * matching their inner provider-level override. Returns `undefined` for
200
+ * providers that should use the shared default.
201
+ */
202
+ export function resolveLazyStreamFirstEventFallbackMs(
203
+ provider: string,
204
+ configuredFallbackMs?: number,
205
+ ): number | undefined {
206
+ if (configuredFallbackMs !== undefined) return configuredFallbackMs;
207
+ return SLOW_FIRST_EVENT_PROVIDERS.has(provider) ? 300_000 : undefined;
208
+ }
193
209
 
194
210
  function forwardStream<TApi extends Api>(
195
211
  target: EventStreamImpl,
@@ -202,11 +218,14 @@ function forwardStream<TApi extends Api>(
202
218
  (async () => {
203
219
  try {
204
220
  const idleTimeoutMs = options.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs(limits?.defaultIdleTimeoutMs);
221
+ const firstEventFallbackMs = resolveLazyStreamFirstEventFallbackMs(
222
+ model.provider,
223
+ limits?.defaultFirstEventTimeoutMs,
224
+ );
205
225
  const watchedSource = iterateWithIdleTimeout(source, {
206
226
  idleTimeoutMs,
207
227
  firstItemTimeoutMs:
208
- options.streamFirstEventTimeoutMs ??
209
- getStreamFirstEventTimeoutMs(idleTimeoutMs, limits?.defaultFirstEventTimeoutMs),
228
+ options.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
210
229
  errorMessage: LAZY_STREAM_IDLE_TIMEOUT_ERROR,
211
230
  firstItemErrorMessage: LAZY_STREAM_FIRST_EVENT_TIMEOUT_ERROR,
212
231
  onIdle: () => abortTracker.abortLocally(new Error(LAZY_STREAM_IDLE_TIMEOUT_ERROR)),
package/src/stream.ts CHANGED
@@ -162,6 +162,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
162
162
  "qwen-portal": () => $pickCredentialEnv("QWEN_OAUTH_TOKEN", "QWEN_PORTAL_API_KEY"),
163
163
  together: "TOGETHER_API_KEY",
164
164
  zenmux: "ZENMUX_API_KEY",
165
+ opengateway: "OPENGATEWAY_API_KEY",
165
166
  venice: "VENICE_API_KEY",
166
167
  vllm: "VLLM_API_KEY",
167
168
  xiaomi: "XIAOMI_API_KEY",
package/src/types.ts CHANGED
@@ -147,6 +147,7 @@ export type KnownProvider =
147
147
  | "minimax"
148
148
  | "opencode-go"
149
149
  | "opencode-zen"
150
+ | "opengateway"
150
151
  | "synthetic"
151
152
  | "cloudflare-ai-gateway"
152
153
  | "huggingface"
@@ -2,6 +2,11 @@ import { $env } from "@gajae-code/utils";
2
2
 
3
3
  const DEFAULT_STREAM_IDLE_TIMEOUT_MS = 120_000;
4
4
  const DEFAULT_STREAM_FIRST_EVENT_TIMEOUT_MS = 100_000;
5
+ const KIMI_CODE_FIRST_EVENT_TIMEOUT_MS = 300_000;
6
+
7
+ export function getProviderFirstEventTimeoutFallbackMs(provider: string): number | undefined {
8
+ return provider === "kimi-code" ? KIMI_CODE_FIRST_EVENT_TIMEOUT_MS : undefined;
9
+ }
5
10
 
6
11
  function normalizeIdleTimeoutMs(value: string | undefined, fallback: number): number | undefined {
7
12
  if (value === undefined) return fallback;
@@ -240,6 +240,11 @@ const builtInOAuthProviders: OAuthProviderInfo[] = [
240
240
  name: "ZenMux",
241
241
  available: true,
242
242
  },
243
+ {
244
+ id: "opengateway",
245
+ name: "OpenGateway by Sionic AI",
246
+ available: true,
247
+ },
243
248
  {
244
249
  id: "vllm",
245
250
  name: "vLLM (Local OpenAI-compatible)",
@@ -386,6 +391,7 @@ export async function refreshOAuthToken(
386
391
  case "vercel-ai-gateway":
387
392
  case "qwen-portal":
388
393
  case "zenmux":
394
+ case "opengateway":
389
395
  case "vllm":
390
396
  // API keys / static bearer tokens don't expire, return as-is
391
397
  newCredentials = credentials;
@@ -0,0 +1,15 @@
1
+ /** OpenGateway (by Sionic AI) login flow (API key paste, validated via /v1/models). */
2
+ import { createApiKeyLogin } from "./api-key-login";
3
+
4
+ export const loginOpenGateway = createApiKeyLogin({
5
+ providerLabel: "OpenGateway by Sionic AI",
6
+ authUrl: "https://opengateway.ai/dashboard",
7
+ instructions: "Create or copy your OpenGateway API key",
8
+ promptMessage: "Paste your OpenGateway API key",
9
+ placeholder: "sk-...",
10
+ validation: {
11
+ kind: "models-endpoint",
12
+ provider: "OpenGateway by Sionic AI",
13
+ modelsUrl: "https://apis.opengateway.ai/v1/models",
14
+ },
15
+ });
@@ -40,6 +40,7 @@ export type OAuthProvider =
40
40
  | "openai-codex-device"
41
41
  | "opencode-go"
42
42
  | "opencode-zen"
43
+ | "opengateway"
43
44
  | "parallel"
44
45
  | "perplexity"
45
46
  | "qianfan"