@gajae-code/ai 0.11.8 → 0.11.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +10 -1
- package/README.md +3 -0
- package/dist/types/provider-models/openai-compat.d.ts +10 -0
- package/dist/types/providers/register-builtins.d.ts +8 -0
- package/dist/types/types.d.ts +1 -1
- package/dist/types/utils/idle-iterator.d.ts +1 -0
- package/dist/types/utils/oauth/opengateway.d.ts +1 -0
- package/dist/types/utils/oauth/types.d.ts +1 -1
- package/package.json +2 -2
- package/src/auth-storage.ts +6 -0
- package/src/cli.ts +1 -0
- package/src/models.json +72 -0
- package/src/provider-models/descriptors.ts +7 -0
- package/src/provider-models/openai-compat.ts +20 -0
- package/src/providers/anthropic.ts +9 -2
- package/src/providers/openai-anthropic-shim.ts +4 -0
- package/src/providers/openai-completions.ts +5 -1
- package/src/providers/register-builtins.ts +21 -2
- package/src/stream.ts +1 -0
- package/src/types.ts +1 -0
- package/src/utils/idle-iterator.ts +5 -0
- package/src/utils/oauth/index.ts +6 -0
- package/src/utils/oauth/opengateway.ts +15 -0
- package/src/utils/oauth/types.ts +1 -0
package/CHANGELOG.md
CHANGED
|
@@ -2,12 +2,21 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [0.11.9] - 2026-07-24
|
|
6
|
+
### Fixed
|
|
7
|
+
|
|
8
|
+
- Kimi Code now allows one continuous 300-second first-event wait before aborting, while preserving explicit caller and environment timeout overrides and the existing inter-event idle timeout.
|
|
9
|
+
|
|
10
|
+
### Added
|
|
11
|
+
|
|
12
|
+
- Added first-class support for **OpenGateway by Sionic AI**, an OpenAI-compatible gateway. Registers the `opengateway` provider descriptor, `/login` OAuth entry (API-key paste validated against `https://apis.opengateway.ai/v1/models`), `OPENGATEWAY_API_KEY` environment resolution, and bundled `models.json` seed models. Models are discovered dynamically from the OpenAI-compatible `/v1/models` endpoint (base URL `https://apis.opengateway.ai/v1`).
|
|
13
|
+
|
|
5
14
|
## [0.11.8] - 2026-07-23
|
|
6
15
|
|
|
7
16
|
### Fixed
|
|
8
17
|
|
|
9
18
|
- OpenAI Responses / Codex native history replay no longer submits missing resident-image placeholders as `input_image.image_url`. Invalid values (including `[Session resident imageUrl blob missing: …]`) are dropped, or retained as `file_id`-only parts when a non-empty `file_id` is present, so a single unavailable historical image cannot brick `/retry` (#2924).
|
|
10
|
-
- Raised the first-event stream timeout floor to five minutes for `alibaba-token-plan` models
|
|
19
|
+
- Raised the first-event stream timeout floor to five minutes for `alibaba-token-plan` models at both the OpenAI provider and outer lazy-stream watchdogs, while preserving caller and environment overrides and the existing inter-event idle timeout.
|
|
11
20
|
- OAuth refresh peer-rotation recovery now runs before failure classification instead of only on the definitive-failure path, and the definitive matcher recognizes the "grant is invalid" phrasing. Providers whose invalid-grant response does not contain the literal `invalid_grant` (e.g. Kimi's 400 "The provided authorization grant is invalid") previously had rotation races misclassified as transient, temp-blocking a healthy credential for five minutes on every race; with Kimi's ~12-minute access tokens and multiple processes sharing the credential store this surfaced as repeated logouts. Genuine revocations are now disabled with a cause instead of looping temp-blocks.
|
|
12
21
|
- Anthropic 400 `Invalid \`signature\` in \`thinking\` block` responses now trigger the one-shot thinking replay repair instead of failing the turn. The existing repair matcher only recognized the "latest assistant message ... cannot be modified" wording, so the signature-validation variant — which can cite a `thinking`/`redacted_thinking` block anywhere in the replayed history (e.g. after compaction/pruning rewrote an earlier turn) — was treated as a fatal request error. The retry now rebuilds the request with thinking blocks dropped from every replayed assistant message (`repairAllAssistantThinking`), while the latest-message mutation variant keeps the targeted latest-only repair.
|
|
13
22
|
|
package/README.md
CHANGED
|
@@ -69,6 +69,7 @@ Unified LLM API with automatic model discovery, provider configuration, token an
|
|
|
69
69
|
- **MiniMax Coding Plan** (requires `MINIMAX_CODE_API_KEY` or `MINIMAX_CODE_CN_API_KEY`)
|
|
70
70
|
- **Xiaomi MiMo** (requires `XIAOMI_API_KEY`)
|
|
71
71
|
- **ZenMux** (requires `ZENMUX_API_KEY`)
|
|
72
|
+
- **OpenGateway by Sionic AI** (requires `OPENGATEWAY_API_KEY`)
|
|
72
73
|
- **Qwen Portal** (supports `QWEN_OAUTH_TOKEN` or `QWEN_PORTAL_API_KEY`)
|
|
73
74
|
- **Cloudflare AI Gateway** (requires `CLOUDFLARE_AI_GATEWAY_API_KEY` and provider-specific gateway base URL)
|
|
74
75
|
- **Ollama** (local OpenAI-compatible runtime; optional `OLLAMA_API_KEY`)
|
|
@@ -954,6 +955,7 @@ In Node.js environments, you can set environment variables to avoid passing API
|
|
|
954
955
|
| MiniMax Code | `MINIMAX_CODE_API_KEY` (international) or `MINIMAX_CODE_CN_API_KEY` (China) |
|
|
955
956
|
| Xiaomi MiMo | `XIAOMI_API_KEY` |
|
|
956
957
|
| ZenMux | `ZENMUX_API_KEY` |
|
|
958
|
+
| OpenGateway | `OPENGATEWAY_API_KEY` |
|
|
957
959
|
| vLLM | `VLLM_API_KEY` |
|
|
958
960
|
| Cloudflare AI Gateway | `CLOUDFLARE_AI_GATEWAY_API_KEY` |
|
|
959
961
|
| GitHub Copilot | `COPILOT_GITHUB_TOKEN` or `GH_TOKEN` or `GITHUB_TOKEN` |
|
|
@@ -977,6 +979,7 @@ Provider endpoint defaults for the current OpenAI-compatible integrations:
|
|
|
977
979
|
- Xiaomi MiMo: `https://api.xiaomimimo.com/anthropic`
|
|
978
980
|
- ZenMux (OpenAI): `https://zenmux.ai/api/v1`
|
|
979
981
|
- ZenMux (Anthropic models): `https://zenmux.ai/api/anthropic`
|
|
982
|
+
- OpenGateway by Sionic AI: `https://apis.opengateway.ai/v1`
|
|
980
983
|
- vLLM: `http://127.0.0.1:8000/v1`
|
|
981
984
|
- Ollama: local OpenAI-compatible runtime (`http://127.0.0.1:11434/v1`)
|
|
982
985
|
- Ollama Cloud: native Ollama API host (`https://ollama.com/api`, configured here as base URL `https://ollama.com`)
|
|
@@ -111,6 +111,16 @@ export interface ZenMuxModelManagerConfig {
|
|
|
111
111
|
baseUrl?: string;
|
|
112
112
|
}
|
|
113
113
|
export declare function zenmuxModelManagerOptions(config?: ZenMuxModelManagerConfig): ModelManagerOptions<Api>;
|
|
114
|
+
export interface OpenGatewayModelManagerConfig {
|
|
115
|
+
apiKey?: string;
|
|
116
|
+
baseUrl?: string;
|
|
117
|
+
}
|
|
118
|
+
/**
|
|
119
|
+
* OpenGateway by Sionic AI — an OpenAI-compatible gateway that fronts OpenAI,
|
|
120
|
+
* Anthropic, and Google models behind one API key. Models are discovered from
|
|
121
|
+
* the OpenAI-compatible `/v1/models` endpoint.
|
|
122
|
+
*/
|
|
123
|
+
export declare function opengatewayModelManagerOptions(config?: OpenGatewayModelManagerConfig): ModelManagerOptions<"openai-completions">;
|
|
114
124
|
export interface KiloModelManagerConfig {
|
|
115
125
|
apiKey?: string;
|
|
116
126
|
baseUrl?: string;
|
|
@@ -17,6 +17,14 @@ interface BedrockProviderModule {
|
|
|
17
17
|
streamBedrock: (model: Model<"bedrock-converse-stream">, context: Context, options: BedrockOptions) => AssistantMessageEventStream;
|
|
18
18
|
}
|
|
19
19
|
export declare function setBedrockProviderModule(module: BedrockProviderModule): void;
|
|
20
|
+
/**
|
|
21
|
+
* Resolves the first-event timeout fallback for the outer lazy-stream watchdog.
|
|
22
|
+
* A configured wrapper-specific fallback (from `LazyStreamLimits`) always wins;
|
|
23
|
+
* otherwise providers known to have slow first events get a five-minute floor
|
|
24
|
+
* matching their inner provider-level override. Returns `undefined` for
|
|
25
|
+
* providers that should use the shared default.
|
|
26
|
+
*/
|
|
27
|
+
export declare function resolveLazyStreamFirstEventFallbackMs(provider: string, configuredFallbackMs?: number): number | undefined;
|
|
20
28
|
export declare const streamAnthropic: (model: Model<"anthropic-messages">, context: Context, options: OptionsForApi<"anthropic-messages">) => EventStreamImpl;
|
|
21
29
|
export declare const streamAzureOpenAIResponses: (model: Model<"azure-openai-responses">, context: Context, options: OptionsForApi<"azure-openai-responses">) => EventStreamImpl;
|
|
22
30
|
export declare const streamGoogle: (model: Model<"google-generative-ai">, context: Context, options: OptionsForApi<"google-generative-ai">) => EventStreamImpl;
|
package/dist/types/types.d.ts
CHANGED
|
@@ -51,7 +51,7 @@ export interface ThinkingConfig {
|
|
|
51
51
|
/** Provider-specific transport used to encode the selected effort. */
|
|
52
52
|
mode: ThinkingControlMode;
|
|
53
53
|
}
|
|
54
|
-
export type KnownProvider = "alibaba-token-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
|
|
54
|
+
export type KnownProvider = "alibaba-token-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "opengateway" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
|
|
55
55
|
export type Provider = KnownProvider | string;
|
|
56
56
|
import type { Effort } from "./model-thinking";
|
|
57
57
|
/** Token budgets for each thinking level (token-based providers only) */
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export declare const loginOpenGateway: (options: import("./types").OAuthController) => Promise<string>;
|
|
@@ -7,7 +7,7 @@ export type OAuthCredentials = {
|
|
|
7
7
|
email?: string;
|
|
8
8
|
accountId?: string;
|
|
9
9
|
};
|
|
10
|
-
export type OAuthProvider = "alibaba-token-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
|
|
10
|
+
export type OAuthProvider = "alibaba-token-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "opengateway" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
|
|
11
11
|
export type OAuthProviderId = OAuthProvider | (string & {});
|
|
12
12
|
export type OAuthPrompt = {
|
|
13
13
|
message: string;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@gajae-code/ai",
|
|
4
|
-
"version": "0.11.
|
|
4
|
+
"version": "0.11.9",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://gajae-code.com",
|
|
7
7
|
"author": "Yeachan-Heo and Gajae Code Contributors",
|
|
@@ -40,7 +40,7 @@
|
|
|
40
40
|
"dependencies": {
|
|
41
41
|
"@anthropic-ai/sdk": "^0.94.0",
|
|
42
42
|
"@bufbuild/protobuf": "^2.12.0",
|
|
43
|
-
"@gajae-code/utils": "0.11.
|
|
43
|
+
"@gajae-code/utils": "0.11.9",
|
|
44
44
|
"openai": "^6.36.0",
|
|
45
45
|
"partial-json": "^0.1.7",
|
|
46
46
|
"zod": "4.4.3"
|
package/src/auth-storage.ts
CHANGED
|
@@ -1998,6 +1998,12 @@ export class AuthStorage {
|
|
|
1998
1998
|
await saveApiKeyCredential(apiKey);
|
|
1999
1999
|
return;
|
|
2000
2000
|
}
|
|
2001
|
+
case "opengateway": {
|
|
2002
|
+
const { loginOpenGateway } = await import("./utils/oauth/opengateway");
|
|
2003
|
+
const apiKey = await loginOpenGateway(ctrl);
|
|
2004
|
+
await saveApiKeyCredential(apiKey);
|
|
2005
|
+
return;
|
|
2006
|
+
}
|
|
2001
2007
|
default: {
|
|
2002
2008
|
const customProvider = getOAuthProvider(provider);
|
|
2003
2009
|
if (!customProvider) {
|
package/src/cli.ts
CHANGED
package/src/models.json
CHANGED
|
@@ -61333,6 +61333,78 @@
|
|
|
61333
61333
|
"maxTokens": 131072
|
|
61334
61334
|
}
|
|
61335
61335
|
},
|
|
61336
|
+
"opengateway": {
|
|
61337
|
+
"openai/gpt-4o": {
|
|
61338
|
+
"id": "openai/gpt-4o",
|
|
61339
|
+
"name": "GPT-4o (OpenGateway)",
|
|
61340
|
+
"api": "openai-completions",
|
|
61341
|
+
"provider": "opengateway",
|
|
61342
|
+
"baseUrl": "https://apis.opengateway.ai/v1",
|
|
61343
|
+
"reasoning": false,
|
|
61344
|
+
"input": [
|
|
61345
|
+
"text",
|
|
61346
|
+
"image"
|
|
61347
|
+
],
|
|
61348
|
+
"cost": {
|
|
61349
|
+
"input": 2.5,
|
|
61350
|
+
"output": 10,
|
|
61351
|
+
"cacheRead": 1.25,
|
|
61352
|
+
"cacheWrite": 0
|
|
61353
|
+
},
|
|
61354
|
+
"contextWindow": 128000,
|
|
61355
|
+
"maxTokens": 16384
|
|
61356
|
+
},
|
|
61357
|
+
"anthropic/claude-sonnet-4-5": {
|
|
61358
|
+
"id": "anthropic/claude-sonnet-4-5",
|
|
61359
|
+
"name": "Anthropic Sonnet 4.5 (OpenGateway)",
|
|
61360
|
+
"api": "openai-completions",
|
|
61361
|
+
"provider": "opengateway",
|
|
61362
|
+
"baseUrl": "https://apis.opengateway.ai/v1",
|
|
61363
|
+
"reasoning": true,
|
|
61364
|
+
"input": [
|
|
61365
|
+
"text",
|
|
61366
|
+
"image"
|
|
61367
|
+
],
|
|
61368
|
+
"cost": {
|
|
61369
|
+
"input": 3,
|
|
61370
|
+
"output": 15,
|
|
61371
|
+
"cacheRead": 0.3,
|
|
61372
|
+
"cacheWrite": 3.75
|
|
61373
|
+
},
|
|
61374
|
+
"contextWindow": 200000,
|
|
61375
|
+
"maxTokens": 64000,
|
|
61376
|
+
"thinking": {
|
|
61377
|
+
"mode": "effort",
|
|
61378
|
+
"minLevel": "minimal",
|
|
61379
|
+
"maxLevel": "high"
|
|
61380
|
+
}
|
|
61381
|
+
},
|
|
61382
|
+
"google/gemini-2.5-pro": {
|
|
61383
|
+
"id": "google/gemini-2.5-pro",
|
|
61384
|
+
"name": "Gemini 2.5 Pro (OpenGateway)",
|
|
61385
|
+
"api": "openai-completions",
|
|
61386
|
+
"provider": "opengateway",
|
|
61387
|
+
"baseUrl": "https://apis.opengateway.ai/v1",
|
|
61388
|
+
"reasoning": true,
|
|
61389
|
+
"input": [
|
|
61390
|
+
"text",
|
|
61391
|
+
"image"
|
|
61392
|
+
],
|
|
61393
|
+
"cost": {
|
|
61394
|
+
"input": 1.25,
|
|
61395
|
+
"output": 10,
|
|
61396
|
+
"cacheRead": 0.31,
|
|
61397
|
+
"cacheWrite": 0
|
|
61398
|
+
},
|
|
61399
|
+
"contextWindow": 1048576,
|
|
61400
|
+
"maxTokens": 65536,
|
|
61401
|
+
"thinking": {
|
|
61402
|
+
"mode": "effort",
|
|
61403
|
+
"minLevel": "minimal",
|
|
61404
|
+
"maxLevel": "high"
|
|
61405
|
+
}
|
|
61406
|
+
}
|
|
61407
|
+
},
|
|
61336
61408
|
"openrouter": {
|
|
61337
61409
|
"~anthropic/claude-fable-latest": {
|
|
61338
61410
|
"id": "~anthropic/claude-fable-latest",
|
|
@@ -33,6 +33,7 @@ import {
|
|
|
33
33
|
openaiModelManagerOptions,
|
|
34
34
|
opencodeGoModelManagerOptions,
|
|
35
35
|
opencodeZenModelManagerOptions,
|
|
36
|
+
opengatewayModelManagerOptions,
|
|
36
37
|
openrouterModelManagerOptions,
|
|
37
38
|
qianfanModelManagerOptions,
|
|
38
39
|
qwenPortalModelManagerOptions,
|
|
@@ -312,6 +313,12 @@ export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [
|
|
|
312
313
|
config => zenmuxModelManagerOptions(config),
|
|
313
314
|
catalog("ZenMux", ["ZENMUX_API_KEY"]),
|
|
314
315
|
),
|
|
316
|
+
catalogDescriptor(
|
|
317
|
+
"opengateway",
|
|
318
|
+
"openai/gpt-4o",
|
|
319
|
+
config => opengatewayModelManagerOptions(config),
|
|
320
|
+
catalog("OpenGateway by Sionic AI", ["OPENGATEWAY_API_KEY"]),
|
|
321
|
+
),
|
|
315
322
|
catalogDescriptor("zai", "glm-5.2", config => zaiModelManagerOptions(config), catalog("zAI", ["ZAI_API_KEY"])),
|
|
316
323
|
catalogDescriptor(
|
|
317
324
|
"glm-zcode",
|
|
@@ -1099,6 +1099,26 @@ export function zenmuxModelManagerOptions(config?: ZenMuxModelManagerConfig): Mo
|
|
|
1099
1099
|
};
|
|
1100
1100
|
}
|
|
1101
1101
|
|
|
1102
|
+
// ---------------------------------------------------------------------------
|
|
1103
|
+
// 10.5.1 OpenGateway by Sionic AI
|
|
1104
|
+
// ---------------------------------------------------------------------------
|
|
1105
|
+
|
|
1106
|
+
export interface OpenGatewayModelManagerConfig {
|
|
1107
|
+
apiKey?: string;
|
|
1108
|
+
baseUrl?: string;
|
|
1109
|
+
}
|
|
1110
|
+
|
|
1111
|
+
/**
|
|
1112
|
+
* OpenGateway by Sionic AI — an OpenAI-compatible gateway that fronts OpenAI,
|
|
1113
|
+
* Anthropic, and Google models behind one API key. Models are discovered from
|
|
1114
|
+
* the OpenAI-compatible `/v1/models` endpoint.
|
|
1115
|
+
*/
|
|
1116
|
+
export function opengatewayModelManagerOptions(
|
|
1117
|
+
config?: OpenGatewayModelManagerConfig,
|
|
1118
|
+
): ModelManagerOptions<"openai-completions"> {
|
|
1119
|
+
return createSimpleOpenAICompletionsOptions("opengateway", "https://apis.opengateway.ai/v1", config);
|
|
1120
|
+
}
|
|
1121
|
+
|
|
1102
1122
|
// ---------------------------------------------------------------------------
|
|
1103
1123
|
// 10.6 Kilo Gateway
|
|
1104
1124
|
// ---------------------------------------------------------------------------
|
|
@@ -60,7 +60,12 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
|
60
60
|
import { transportFailureFacts } from "../utils/fallback-transport";
|
|
61
61
|
import { isFoundryEnabled } from "../utils/foundry";
|
|
62
62
|
import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } from "../utils/http-inspector";
|
|
63
|
-
import {
|
|
63
|
+
import {
|
|
64
|
+
getProviderFirstEventTimeoutFallbackMs,
|
|
65
|
+
getStreamFirstEventTimeoutMs,
|
|
66
|
+
getStreamIdleTimeoutMs,
|
|
67
|
+
iterateWithIdleTimeout,
|
|
68
|
+
} from "../utils/idle-iterator";
|
|
64
69
|
import { parseJsonWithRepair, parseStreamingJson } from "../utils/json-parse";
|
|
65
70
|
import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
|
|
66
71
|
import { notifyProviderResponse } from "../utils/provider-response";
|
|
@@ -1416,7 +1421,9 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1416
1421
|
firstTokenTime = undefined;
|
|
1417
1422
|
};
|
|
1418
1423
|
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs();
|
|
1419
|
-
const
|
|
1424
|
+
const firstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
|
|
1425
|
+
const firstEventTimeoutMs =
|
|
1426
|
+
options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs);
|
|
1420
1427
|
stream.push({ type: "start", partial: output });
|
|
1421
1428
|
// Retry loop for transient errors from the stream.
|
|
1422
1429
|
// Provider-level transport/rate-limit failures: only before any streamed content starts.
|
|
@@ -89,6 +89,8 @@ export function streamOpenAIAnthropicShim(
|
|
|
89
89
|
onResponse: options?.onResponse,
|
|
90
90
|
onSseEvent: options?.onSseEvent,
|
|
91
91
|
fetch: options?.fetch,
|
|
92
|
+
streamIdleTimeoutMs: options?.streamIdleTimeoutMs,
|
|
93
|
+
streamFirstEventTimeoutMs: options?.streamFirstEventTimeoutMs,
|
|
92
94
|
thinkingEnabled,
|
|
93
95
|
thinkingBudgetTokens: thinkingBudget,
|
|
94
96
|
});
|
|
@@ -118,6 +120,8 @@ export function streamOpenAIAnthropicShim(
|
|
|
118
120
|
onResponse: options?.onResponse,
|
|
119
121
|
onSseEvent: options?.onSseEvent,
|
|
120
122
|
fetch: options?.fetch,
|
|
123
|
+
streamIdleTimeoutMs: options?.streamIdleTimeoutMs,
|
|
124
|
+
streamFirstEventTimeoutMs: options?.streamFirstEventTimeoutMs,
|
|
121
125
|
reasoning: reasoningEffort,
|
|
122
126
|
});
|
|
123
127
|
|
|
@@ -49,6 +49,7 @@ import {
|
|
|
49
49
|
import {
|
|
50
50
|
createWatchdog,
|
|
51
51
|
getOpenAIStreamIdleTimeoutMs,
|
|
52
|
+
getProviderFirstEventTimeoutFallbackMs,
|
|
52
53
|
getStreamFirstEventTimeoutMs,
|
|
53
54
|
iterateWithIdleTimeout,
|
|
54
55
|
} from "../utils/idle-iterator";
|
|
@@ -427,6 +428,7 @@ function getTrailingPartialDeepseekToken(text: string): string {
|
|
|
427
428
|
}
|
|
428
429
|
|
|
429
430
|
const ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS = 300_000;
|
|
431
|
+
|
|
430
432
|
const OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE =
|
|
431
433
|
"OpenAI completions stream timed out while waiting for the first event";
|
|
432
434
|
|
|
@@ -564,7 +566,9 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
564
566
|
}
|
|
565
567
|
}
|
|
566
568
|
const firstEventFallbackMs =
|
|
567
|
-
model.provider === "alibaba-token-plan"
|
|
569
|
+
model.provider === "alibaba-token-plan"
|
|
570
|
+
? ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS
|
|
571
|
+
: getProviderFirstEventTimeoutFallbackMs(model.provider);
|
|
568
572
|
const firstEventWatchdog = createWatchdog(
|
|
569
573
|
options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
|
|
570
574
|
() => abortTracker.abortLocally(firstEventTimeoutAbortError),
|
|
@@ -190,6 +190,22 @@ interface LazyStreamLimits {
|
|
|
190
190
|
const GOOGLE_GEMINI_CLI_LAZY_STREAM_LIMITS: LazyStreamLimits = {
|
|
191
191
|
defaultFirstEventTimeoutMs: 300_000,
|
|
192
192
|
};
|
|
193
|
+
const SLOW_FIRST_EVENT_PROVIDERS = new Set(["alibaba-token-plan", "kimi-code"]);
|
|
194
|
+
|
|
195
|
+
/**
|
|
196
|
+
* Resolves the first-event timeout fallback for the outer lazy-stream watchdog.
|
|
197
|
+
* A configured wrapper-specific fallback (from `LazyStreamLimits`) always wins;
|
|
198
|
+
* otherwise providers known to have slow first events get a five-minute floor
|
|
199
|
+
* matching their inner provider-level override. Returns `undefined` for
|
|
200
|
+
* providers that should use the shared default.
|
|
201
|
+
*/
|
|
202
|
+
export function resolveLazyStreamFirstEventFallbackMs(
|
|
203
|
+
provider: string,
|
|
204
|
+
configuredFallbackMs?: number,
|
|
205
|
+
): number | undefined {
|
|
206
|
+
if (configuredFallbackMs !== undefined) return configuredFallbackMs;
|
|
207
|
+
return SLOW_FIRST_EVENT_PROVIDERS.has(provider) ? 300_000 : undefined;
|
|
208
|
+
}
|
|
193
209
|
|
|
194
210
|
function forwardStream<TApi extends Api>(
|
|
195
211
|
target: EventStreamImpl,
|
|
@@ -202,11 +218,14 @@ function forwardStream<TApi extends Api>(
|
|
|
202
218
|
(async () => {
|
|
203
219
|
try {
|
|
204
220
|
const idleTimeoutMs = options.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs(limits?.defaultIdleTimeoutMs);
|
|
221
|
+
const firstEventFallbackMs = resolveLazyStreamFirstEventFallbackMs(
|
|
222
|
+
model.provider,
|
|
223
|
+
limits?.defaultFirstEventTimeoutMs,
|
|
224
|
+
);
|
|
205
225
|
const watchedSource = iterateWithIdleTimeout(source, {
|
|
206
226
|
idleTimeoutMs,
|
|
207
227
|
firstItemTimeoutMs:
|
|
208
|
-
options.streamFirstEventTimeoutMs ??
|
|
209
|
-
getStreamFirstEventTimeoutMs(idleTimeoutMs, limits?.defaultFirstEventTimeoutMs),
|
|
228
|
+
options.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
|
|
210
229
|
errorMessage: LAZY_STREAM_IDLE_TIMEOUT_ERROR,
|
|
211
230
|
firstItemErrorMessage: LAZY_STREAM_FIRST_EVENT_TIMEOUT_ERROR,
|
|
212
231
|
onIdle: () => abortTracker.abortLocally(new Error(LAZY_STREAM_IDLE_TIMEOUT_ERROR)),
|
package/src/stream.ts
CHANGED
|
@@ -162,6 +162,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
|
|
|
162
162
|
"qwen-portal": () => $pickCredentialEnv("QWEN_OAUTH_TOKEN", "QWEN_PORTAL_API_KEY"),
|
|
163
163
|
together: "TOGETHER_API_KEY",
|
|
164
164
|
zenmux: "ZENMUX_API_KEY",
|
|
165
|
+
opengateway: "OPENGATEWAY_API_KEY",
|
|
165
166
|
venice: "VENICE_API_KEY",
|
|
166
167
|
vllm: "VLLM_API_KEY",
|
|
167
168
|
xiaomi: "XIAOMI_API_KEY",
|
package/src/types.ts
CHANGED
|
@@ -2,6 +2,11 @@ import { $env } from "@gajae-code/utils";
|
|
|
2
2
|
|
|
3
3
|
const DEFAULT_STREAM_IDLE_TIMEOUT_MS = 120_000;
|
|
4
4
|
const DEFAULT_STREAM_FIRST_EVENT_TIMEOUT_MS = 100_000;
|
|
5
|
+
const KIMI_CODE_FIRST_EVENT_TIMEOUT_MS = 300_000;
|
|
6
|
+
|
|
7
|
+
export function getProviderFirstEventTimeoutFallbackMs(provider: string): number | undefined {
|
|
8
|
+
return provider === "kimi-code" ? KIMI_CODE_FIRST_EVENT_TIMEOUT_MS : undefined;
|
|
9
|
+
}
|
|
5
10
|
|
|
6
11
|
function normalizeIdleTimeoutMs(value: string | undefined, fallback: number): number | undefined {
|
|
7
12
|
if (value === undefined) return fallback;
|
package/src/utils/oauth/index.ts
CHANGED
|
@@ -240,6 +240,11 @@ const builtInOAuthProviders: OAuthProviderInfo[] = [
|
|
|
240
240
|
name: "ZenMux",
|
|
241
241
|
available: true,
|
|
242
242
|
},
|
|
243
|
+
{
|
|
244
|
+
id: "opengateway",
|
|
245
|
+
name: "OpenGateway by Sionic AI",
|
|
246
|
+
available: true,
|
|
247
|
+
},
|
|
243
248
|
{
|
|
244
249
|
id: "vllm",
|
|
245
250
|
name: "vLLM (Local OpenAI-compatible)",
|
|
@@ -386,6 +391,7 @@ export async function refreshOAuthToken(
|
|
|
386
391
|
case "vercel-ai-gateway":
|
|
387
392
|
case "qwen-portal":
|
|
388
393
|
case "zenmux":
|
|
394
|
+
case "opengateway":
|
|
389
395
|
case "vllm":
|
|
390
396
|
// API keys / static bearer tokens don't expire, return as-is
|
|
391
397
|
newCredentials = credentials;
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
/** OpenGateway (by Sionic AI) login flow (API key paste, validated via /v1/models). */
|
|
2
|
+
import { createApiKeyLogin } from "./api-key-login";
|
|
3
|
+
|
|
4
|
+
export const loginOpenGateway = createApiKeyLogin({
|
|
5
|
+
providerLabel: "OpenGateway by Sionic AI",
|
|
6
|
+
authUrl: "https://opengateway.ai/dashboard",
|
|
7
|
+
instructions: "Create or copy your OpenGateway API key",
|
|
8
|
+
promptMessage: "Paste your OpenGateway API key",
|
|
9
|
+
placeholder: "sk-...",
|
|
10
|
+
validation: {
|
|
11
|
+
kind: "models-endpoint",
|
|
12
|
+
provider: "OpenGateway by Sionic AI",
|
|
13
|
+
modelsUrl: "https://apis.opengateway.ai/v1/models",
|
|
14
|
+
},
|
|
15
|
+
});
|