@gajae-code/ai 0.11.11 → 0.12.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +42 -0
- package/README.md +3 -0
- package/dist/types/model-thinking.d.ts +15 -0
- package/dist/types/provider-models/openai-compat.d.ts +7 -0
- package/dist/types/providers/anthropic.d.ts +2 -1
- package/dist/types/providers/azure-openai-responses.d.ts +8 -1
- package/dist/types/providers/google-auth.d.ts +2 -0
- package/dist/types/providers/google-gemini-headers.d.ts +1 -1
- package/dist/types/providers/google-vertex.d.ts +4 -0
- package/dist/types/providers/openai-codex-responses.d.ts +4 -0
- package/dist/types/providers/openai-completions.d.ts +2 -0
- package/dist/types/providers/openai-responses.d.ts +2 -0
- package/dist/types/types.d.ts +1 -1
- package/dist/types/usage/grok-cli.d.ts +3 -1
- package/dist/types/usage/kimi.d.ts +2 -0
- package/dist/types/utils/anthropic-auth.d.ts +8 -0
- package/dist/types/utils/fallback-transport.d.ts +2 -0
- package/dist/types/utils/foundry.d.ts +10 -0
- package/dist/types/utils/http-inspector.d.ts +23 -0
- package/dist/types/utils/idle-iterator.d.ts +4 -0
- package/dist/types/utils/oauth/bizrouter.d.ts +1 -0
- package/dist/types/utils/oauth/kimi.d.ts +2 -0
- package/dist/types/utils/oauth/perplexity.d.ts +2 -7
- package/dist/types/utils/oauth/types.d.ts +1 -1
- package/package.json +2 -2
- package/src/auth-storage.ts +6 -0
- package/src/cli.ts +1 -0
- package/src/model-thinking.ts +19 -0
- package/src/models.json +72 -0
- package/src/provider-models/descriptors.ts +7 -0
- package/src/provider-models/openai-compat.ts +67 -6
- package/src/providers/amazon-bedrock.ts +5 -20
- package/src/providers/anthropic.ts +103 -48
- package/src/providers/azure-openai-responses.ts +44 -19
- package/src/providers/google-auth.ts +13 -2
- package/src/providers/google-gemini-headers.ts +1 -1
- package/src/providers/google-vertex.ts +41 -5
- package/src/providers/ollama.ts +32 -4
- package/src/providers/openai-codex-responses.ts +97 -38
- package/src/providers/openai-completions.ts +25 -14
- package/src/providers/openai-responses.ts +22 -20
- package/src/providers/register-builtins.ts +12 -2
- package/src/stream.ts +1 -0
- package/src/types.ts +1 -0
- package/src/usage/claude.ts +2 -1
- package/src/usage/grok-cli.ts +12 -1
- package/src/usage/kimi.ts +16 -2
- package/src/utils/anthropic-auth.ts +11 -3
- package/src/utils/fallback-transport.ts +17 -10
- package/src/utils/foundry.ts +12 -2
- package/src/utils/http-inspector.ts +124 -1
- package/src/utils/idle-iterator.ts +12 -3
- package/src/utils/oauth/bizrouter.ts +15 -0
- package/src/utils/oauth/index.ts +6 -0
- package/src/utils/oauth/kimi.ts +17 -2
- package/src/utils/oauth/perplexity.ts +21 -2
- package/src/utils/oauth/types.ts +1 -0
- package/src/utils/schema/adapt.ts +2 -2
- package/src/utils/tool-choice-capability.ts +2 -1
package/CHANGELOG.md
CHANGED
|
@@ -2,12 +2,54 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [0.12.1] - 2026-07-29
|
|
6
|
+
|
|
7
|
+
### Fixed
|
|
8
|
+
|
|
9
|
+
- Lazy-stream first-event timeouts now abort with `FirstEventTimeoutError` so `transportFailure.providerCode` is `stream_first_event_timeout` on the outer watchdog path shared by all bundled providers via `createLazyStream`. Idle stalls remain bare `Error`s (distinct class intentionally) (#3496).
|
|
10
|
+
|
|
11
|
+
- Provider streams now surface first-event watchdog expiry as a typed timeout so callers can apply bounded retry policy without parsing error prose.
|
|
12
|
+
- Codex websocket first-event timeouts now discard the timed-out connection before the outer retry/fallback layer handles the typed failure, preventing late frames from the abandoned request from being consumed by the replayed turn.
|
|
13
|
+
- Codex named-tool requests now recognize provider `Tool choice '<name>' not found in 'tools' parameter` errors as runtime capability failures and retry once without forcing the choice.
|
|
14
|
+
- The Kimi OAuth host (`KIMI_CODE_OAUTH_HOST` / `KIMI_OAUTH_HOST`) is now resolved from trusted environment sources only. That host receives the device-authorization request, the authorization-code exchange, and the refresh call that carries the existing refresh token, so reading it through the merged view that includes the caller's `cwd/.env` let a repository redirect the login flow and collect the user's Kimi credentials. Resolution now uses the non-project resolver; shell and user-level configuration is unchanged.
|
|
15
|
+
- The documented `GJC_NO_STRICT` environment variable now takes effect. `adaptSchemaForStrict` read only the legacy `PI_NO_STRICT`, so an operator hitting a provider that rejects strict function schemas set the documented name and strict mode stayed on. Both names are honoured, canonical name first, and `GJC_NO_STRICT` is now listed in the environment-variable reference rather than only in the schema-normalisation note.
|
|
16
|
+
- The documented `GJC_AUTH_NO_BORROW` environment variable now takes effect. Only the legacy `PI_AUTH_NO_BORROW` was read, so an operator who followed the documentation to disable macOS native-app token borrowing still had a JWT read out of the Perplexity desktop application during login. Both names are now honoured, and the contract stays presence-based as documented so that setting it to `0` cannot silently re-enable borrowing.
|
|
17
|
+
- The Azure client's `AZURE_OPENAI_API_KEY` fallback is now resolved from trusted environment sources only. It read the merged view that includes the caller's `cwd/.env`, so a repository could supply the credential the client authenticates with; provider credential resolution is documented as excluding the project `.env`, and this fallback now matches. An explicit caller-supplied key still takes precedence, and shell / user-level configuration is unchanged.
|
|
18
|
+
- Anthropic and Ollama tool calls cut off by an output-token limit are now marked incomplete before dispatch, so repaired partial JSON is rejected instead of executing with truncated arguments.
|
|
19
|
+
- The Anthropic "thinking blocks in the latest assistant message cannot be modified" 400 now escalates its one-shot replay repair. The error names the latest assistant message but its cited `messages.N.content.M` path can point at an earlier replayed turn, so the latest-only repair was rejected identically and killed the turn; recovery now retries once more with thinking dropped from every replayed assistant message.
|
|
20
|
+
- Anthropic adaptive-thinking `display` support is now decided by the canonical model-version parser instead of a provider-local `claude-opus-(\d+)-(\d+)` regex. The regex only matched two-component ids, so a single-component alias such as `claude-opus-5` was classified as pre-4.7 while its dated snapshot `claude-opus-5-20260101` was not: the alias sent `thinking: { type: "adaptive" }` without `display: "summarized"`, additionally requested the `interleaved-thinking-2025-05-14` beta, and had its returned thinking blocks recorded as raw rather than summarized. Both Anthropic and Bedrock providers now share `supportsAnthropicAdaptiveThinkingDisplay`, so alias and dated ids of the same model send an identical request shape.
|
|
21
|
+
- Anthropic requests that force a tool choice no longer replay signed thinking blocks. Forcing `tool_choice` strips `thinking` from the request (the API rejects the combination), but the converted history still carried native `thinking`/`redacted_thinking` blocks from thinking-enabled turns, so eager tool-forcing turns (e.g. the todo bootstrap) sent a request whose history contradicted its own thinking setting and drew a 400. The replay now degrades in the same rebuild; the forced request trades its prompt-cache prefix for a shape the API accepts.
|
|
22
|
+
|
|
23
|
+
## [0.12.0] - 2026-07-28
|
|
24
|
+
|
|
25
|
+
### Added
|
|
26
|
+
|
|
27
|
+
- Added first-class support for **BizRouter**, an OpenAI-compatible Korean enterprise LLM gateway. Registers the `bizrouter` provider descriptor, `/login` entry (API-key paste validated against `https://api.bizrouter.ai/v1/models`), `BIZROUTER_API_KEY` environment resolution, and bundled `models.json` seed models. Models are discovered dynamically from `GET /v1/models` (base URL `https://api.bizrouter.ai/v1`).
|
|
28
|
+
### Fixed
|
|
29
|
+
|
|
30
|
+
- Anthropic subscription OAuth requests now use the current Claude Code compatibility attribution (`2.1.219`, `sdk-cli`) instead of the stale `2.1.63` CLI fingerprint that Anthropic can misclassify as extra usage.
|
|
31
|
+
- Connection failures now name the transport code and the target URL. Bun reports DNS and socket failures as a bare `Error` whose message is a standalone hint ("Was there a typo in the url or port?", "Unable to connect. Is the computer able to access the url?") and keeps the actionable facts on `code` and `path`, but only `message` reached the assistant message. A provider outage, a local DNS failure, and a mistyped custom base URL therefore all rendered as the same context-free sentence with no host in it. Such failures now read `... (transport=FailedToOpenSocket url=https://chatgpt.com/backend-api/codex/responses)`; the URL is reduced to origin and path so a key carried in the query string is not surfaced.
|
|
32
|
+
|
|
33
|
+
### Documentation
|
|
34
|
+
|
|
35
|
+
- `docs/environment-variables.md` now names the Anthropic Foundry gateway variables that are actually read: `CLAUDE_CODE_USE_FOUNDRY`, `CLAUDE_CODE_CLIENT_CERT`, and `CLAUDE_CODE_CLIENT_KEY`. The page advertised `ANTHROPIC_MODEL_CODE_*` spellings that no code path reads, so an operator following it could not enable Foundry mode at all, and the mTLS client material was silently ignored.
|
|
36
|
+
|
|
5
37
|
## [0.11.11] - 2026-07-26
|
|
6
38
|
|
|
7
39
|
### Fixed
|
|
8
40
|
|
|
41
|
+
- The Kimi usage endpoint base (`KIMI_CODE_BASE_URL`) is now resolved from trusted environment sources only. That base becomes the URL the usage request sends `Authorization: Bearer <accessToken>` to, so reading it through the merged view that includes the caller's `cwd/.env` let a repository collect the user's Kimi access token. An explicit caller-supplied base URL still takes precedence, and shell / user-level configuration is unchanged.
|
|
42
|
+
- The Gemini CLI compatibility version used in the outbound `User-Agent` is refreshed from `0.50.0` to `0.52.0`, matching the current upstream release. The repository ships `check-spoofed-versions` for exactly this, but that check is not wired into CI, so the value had drifted two minor releases behind.
|
|
43
|
+
- The OpenAI and Azure endpoint decisions are now resolved from trusted environment sources only: `OPENAI_BASE_URL` (streaming responses, completions, and the model manager) and `AZURE_OPENAI_BASE_URL`. `Bun.env` is `process.env` and the env module merges the caller's `cwd/.env` into it, so a repository could previously plant a `.env` that redirected authenticated requests; the two provider paths already reached for `$inheritedEnv` but re-admitted the project `.env` through a fallback. Resolution now goes through the non-project resolver (launching shell plus GJC/user-owned `.env` files); shell and user-level configuration is unchanged.
|
|
44
|
+
- The OpenAI and Azure endpoint decisions are now resolved from trusted environment sources only: `OPENAI_BASE_URL` (streaming responses, completions, and the model manager), `AZURE_OPENAI_BASE_URL`, and `AZURE_OPENAI_RESOURCE_NAME` (the alternate constructor for the same Azure host). `Bun.env` is `process.env` and the env module merges the caller's `cwd/.env` into it, so a repository could previously plant a `.env` that redirected authenticated requests; the two provider paths already reached for `$inheritedEnv` but re-admitted the project `.env` through a fallback. Resolution now goes through the non-project resolver (launching shell plus GJC/user-owned `.env` files); shell and user-level configuration is unchanged.
|
|
45
|
+
- Google credential material is now resolved from trusted environment sources only: the `GOOGLE_APPLICATION_CREDENTIALS` service-account / authorized-user file path used by the ADC loader, and `GOOGLE_CLOUD_API_KEY` used as the Vertex API key. Both were read through the merged view that includes the caller's `cwd/.env`, so a repository could ship a key file and point the agent at it, making it authenticate to Google as an identity the repository chose. `stream.ts` already resolved the same ADC variable through the non-project resolver; the two now agree. An explicit caller-supplied API key still takes precedence.
|
|
46
|
+
- The Grok usage token fallback (`GROK_CLI_OAUTH_TOKEN`) is now resolved from trusted environment sources only. It authenticates the billing/usage call, and reading it through the merged view that includes the caller's `cwd/.env` let a repository decide which account that call ran against. Stored credentials keep precedence, and shell / user-level configuration is unchanged.
|
|
47
|
+
- The Vertex AI location (`GOOGLE_CLOUD_LOCATION`) can no longer redirect authenticated requests off Google. It is interpolated into the request host (`${location}-aiplatform.googleapis.com`), so a value containing `/` terminated the authority component and sent the Google access token to an arbitrary origin — and it was read through the merged view that includes the caller's `cwd/.env`. It now resolves from trusted sources only and must be a region label; `GOOGLE_CLOUD_PROJECT` / `GCLOUD_PROJECT` moved to the same trusted resolver.
|
|
48
|
+
- HTTP 400 request dumps are now bounded. Every 400 wrote a file containing the full sanitized request body and nothing ever removed one, so the directory grew without limit — a developer machine reached 27,249 files totalling 7.0 GB, which was 96% of everything under `~/.gjc`. The newest 50 are retained, matching the bounded retention the rotating application log already uses, and pruning stays best-effort so diagnostics never turn a request failure into a second failure.
|
|
9
49
|
- Anthropic `ping` keepalives no longer reset stream progress, so responses that stop producing content now reach the idle timeout instead of hanging indefinitely.
|
|
50
|
+
- The Anthropic endpoint decision is now resolved from trusted environment sources only: `ANTHROPIC_BASE_URL`, `FOUNDRY_BASE_URL`, `ZCODE_PLAN_ANTHROPIC_BASE_URL`, and the `CLAUDE_CODE_USE_FOUNDRY` mode switch. `Bun.env` is `process.env` and the env module merges the caller's `cwd/.env` into it, so a repository could previously plant a `.env` that redirected authenticated Anthropic requests — the resolved base URL becomes `${baseUrl}/v1/messages` while the headers carry the API key or OAuth token. Resolution now goes through the non-project resolver (launching shell plus GJC/user-owned `.env` files); shell and user-level configuration is unchanged.
|
|
10
51
|
- The documented `GJC_OPENAI_STREAM_IDLE_TIMEOUT_MS` environment variable now takes effect: the stream-watchdog idle-timeout helpers resolve it GJC-first before the legacy `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` / `PI_STREAM_IDLE_TIMEOUT_MS` aliases (previously only the `PI_`-prefixed names were read, so setting the documented GJC name was a silent no-op).
|
|
52
|
+
- The documented OpenAI-code provider knobs now take effect: `GJC_OPENAI_CODE_DEBUG`, `GJC_OPENAI_CODE_WEBSOCKET`, `GJC_OPENAI_CODE_WEBSOCKET_IDLE_TIMEOUT_MS`, `GJC_OPENAI_CODE_WEBSOCKET_RETRY_BUDGET`, and `GJC_OPENAI_CODE_WEBSOCKET_RETRY_DELAY_MS` are resolved GJC-first ahead of the legacy `PI_CODEX_*` names. The Codex → OpenAI-code rename had updated the documentation but not the reads, so every documented name was a silent no-op.
|
|
11
53
|
|
|
12
54
|
## [0.11.9] - 2026-07-24
|
|
13
55
|
### Fixed
|
package/README.md
CHANGED
|
@@ -70,6 +70,7 @@ Unified LLM API with automatic model discovery, provider configuration, token an
|
|
|
70
70
|
- **Xiaomi MiMo** (requires `XIAOMI_API_KEY`)
|
|
71
71
|
- **ZenMux** (requires `ZENMUX_API_KEY`)
|
|
72
72
|
- **OpenGateway by Sionic AI** (requires `OPENGATEWAY_API_KEY`)
|
|
73
|
+
- **BizRouter** (requires `BIZROUTER_API_KEY`)
|
|
73
74
|
- **Qwen Portal** (supports `QWEN_OAUTH_TOKEN` or `QWEN_PORTAL_API_KEY`)
|
|
74
75
|
- **Cloudflare AI Gateway** (requires `CLOUDFLARE_AI_GATEWAY_API_KEY` and provider-specific gateway base URL)
|
|
75
76
|
- **Ollama** (local OpenAI-compatible runtime; optional `OLLAMA_API_KEY`)
|
|
@@ -956,6 +957,7 @@ In Node.js environments, you can set environment variables to avoid passing API
|
|
|
956
957
|
| Xiaomi MiMo | `XIAOMI_API_KEY` |
|
|
957
958
|
| ZenMux | `ZENMUX_API_KEY` |
|
|
958
959
|
| OpenGateway | `OPENGATEWAY_API_KEY` |
|
|
960
|
+
| BizRouter | `BIZROUTER_API_KEY` |
|
|
959
961
|
| vLLM | `VLLM_API_KEY` |
|
|
960
962
|
| Cloudflare AI Gateway | `CLOUDFLARE_AI_GATEWAY_API_KEY` |
|
|
961
963
|
| GitHub Copilot | `COPILOT_GITHUB_TOKEN` or `GH_TOKEN` or `GITHUB_TOKEN` |
|
|
@@ -980,6 +982,7 @@ Provider endpoint defaults for the current OpenAI-compatible integrations:
|
|
|
980
982
|
- ZenMux (OpenAI): `https://zenmux.ai/api/v1`
|
|
981
983
|
- ZenMux (Anthropic models): `https://zenmux.ai/api/anthropic`
|
|
982
984
|
- OpenGateway by Sionic AI: `https://apis.opengateway.ai/v1`
|
|
985
|
+
- BizRouter: `https://api.bizrouter.ai/v1`
|
|
983
986
|
- vLLM: `http://127.0.0.1:8000/v1`
|
|
984
987
|
- Ollama: local OpenAI-compatible runtime (`http://127.0.0.1:11434/v1`)
|
|
985
988
|
- Ollama Cloud: native Ollama API host (`https://ollama.com/api`, configured here as base URL `https://ollama.com`)
|
|
@@ -72,3 +72,18 @@ export declare function mapEffortToAnthropicAdaptiveEffort<TApi extends Api>(mod
|
|
|
72
72
|
* - Thinking content is omitted by default (needs display: "summarized")
|
|
73
73
|
*/
|
|
74
74
|
export declare function hasOpus47ApiRestrictions(modelId: string): boolean;
|
|
75
|
+
/**
|
|
76
|
+
* Adaptive thinking `display` is supported starting with Anthropic Opus 4.7.
|
|
77
|
+
* Older adaptive-thinking models (Opus 4.6, Sonnet 4.6+) reject the field.
|
|
78
|
+
* Fable (5+) postdates Opus 4.7, accepts `display`, and defaults it to
|
|
79
|
+
* "omitted" — thinking tokens are billed but no content streams back — so it
|
|
80
|
+
* must opt in like Opus 4.7+ (issue #2791).
|
|
81
|
+
*
|
|
82
|
+
* Shares `hasOpus47ApiRestrictions` version parsing on purpose: the two
|
|
83
|
+
* predicates describe the same API generation, and a private `claude-opus-(\d+)-(\d+)`
|
|
84
|
+
* regex silently disagreed with it for single-component aliases (`claude-opus-5`
|
|
85
|
+
* matched nothing while `claude-opus-5-20260101` matched), so the same model sent
|
|
86
|
+
* a different thinking shape and beta set depending on which id string was used.
|
|
87
|
+
* Bedrock region/inference-profile prefixes are handled by the canonical parser.
|
|
88
|
+
*/
|
|
89
|
+
export declare function supportsAnthropicAdaptiveThinkingDisplay(modelId: string): boolean;
|
|
@@ -27,6 +27,8 @@ export interface OpenAIModelManagerConfig {
|
|
|
27
27
|
apiKey?: string;
|
|
28
28
|
baseUrl?: string;
|
|
29
29
|
}
|
|
30
|
+
/** Test seam: the model-manager base URL as resolved from trusted env. */
|
|
31
|
+
export declare function resolveOpenAIModelManagerBaseUrlForTest(config?: OpenAIModelManagerConfig): string;
|
|
30
32
|
export declare function openaiModelManagerOptions(config?: OpenAIModelManagerConfig): ModelManagerOptions<"openai-responses">;
|
|
31
33
|
export interface GroqModelManagerConfig {
|
|
32
34
|
apiKey?: string;
|
|
@@ -121,6 +123,11 @@ export interface OpenGatewayModelManagerConfig {
|
|
|
121
123
|
* the OpenAI-compatible `/v1/models` endpoint.
|
|
122
124
|
*/
|
|
123
125
|
export declare function opengatewayModelManagerOptions(config?: OpenGatewayModelManagerConfig): ModelManagerOptions<"openai-completions">;
|
|
126
|
+
export interface BizRouterModelManagerConfig {
|
|
127
|
+
apiKey?: string;
|
|
128
|
+
baseUrl?: string;
|
|
129
|
+
}
|
|
130
|
+
export declare function bizrouterModelManagerOptions(config?: BizRouterModelManagerConfig): ModelManagerOptions<"openai-completions">;
|
|
124
131
|
export interface KiloModelManagerConfig {
|
|
125
132
|
apiKey?: string;
|
|
126
133
|
baseUrl?: string;
|
|
@@ -50,7 +50,8 @@ export declare function isAnthropicThinkingBlockMutationError(error: unknown): b
|
|
|
50
50
|
* than only the latest one.
|
|
51
51
|
*/
|
|
52
52
|
export declare function isAnthropicThinkingSignatureInvalidError(error: unknown): boolean;
|
|
53
|
-
export declare const claudeCodeVersion = "2.1.
|
|
53
|
+
export declare const claudeCodeVersion = "2.1.219";
|
|
54
|
+
export declare const claudeCodeEntrypoint = "sdk-cli";
|
|
54
55
|
export declare const claudeToolPrefix: string;
|
|
55
56
|
export declare const claudeCodeSystemInstruction = "You are a Claude agent, built on Anthropic's Claude Agent SDK.";
|
|
56
57
|
export declare function mapStainlessOs(platform: string): "MacOS" | "Windows" | "Linux" | "FreeBSD" | `Other::${string}`;
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { ServiceTier, StreamFunction, StreamOptions, ToolChoice } from "../types";
|
|
1
|
+
import type { Model, ServiceTier, StreamFunction, StreamOptions, ToolChoice } from "../types";
|
|
2
2
|
export interface AzureOpenAIResponsesOptions extends StreamOptions {
|
|
3
3
|
reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
|
|
4
4
|
reasoningSummary?: "auto" | "detailed" | "concise" | null;
|
|
@@ -13,3 +13,10 @@ export interface AzureOpenAIResponsesOptions extends StreamOptions {
|
|
|
13
13
|
* Generate function for Azure OpenAI Responses API
|
|
14
14
|
*/
|
|
15
15
|
export declare const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses">;
|
|
16
|
+
/** Test seam: the Azure endpoint config as resolved from trusted env. */
|
|
17
|
+
export declare function resolveAzureConfigForTest(model: Model<"azure-openai-responses">, options?: AzureOpenAIResponsesOptions): {
|
|
18
|
+
baseUrl: string;
|
|
19
|
+
apiVersion: string;
|
|
20
|
+
};
|
|
21
|
+
/** Test seam: the client API key as resolved from a caller value plus trusted env. */
|
|
22
|
+
export declare function resolveAzureClientApiKeyForTest(apiKey: string): string | undefined;
|
|
@@ -12,6 +12,8 @@
|
|
|
12
12
|
* (default 60s). Concurrent callers waiting on a refresh share the same in-flight promise.
|
|
13
13
|
*/
|
|
14
14
|
import type { FetchImpl } from "../types";
|
|
15
|
+
/** Test seam: the ADC credentials file path as resolved from trusted env. */
|
|
16
|
+
export declare function resolveAdcCredentialsPathForTest(): string | undefined;
|
|
15
17
|
/**
|
|
16
18
|
* Returns a Bearer access token suitable for the `Authorization` header on Vertex AI calls.
|
|
17
19
|
* The token is cached in module scope and refreshed `GOOGLE_VERTEX_REFRESH_SKEW_MS` ms before it expires.
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
*/
|
|
6
6
|
export declare const GEMINI_CLI_VERSION_ENV = "GJC_AI_GEMINI_CLI_VERSION";
|
|
7
7
|
export declare const LEGACY_GEMINI_CLI_VERSION_ENV = "PI_AI_GEMINI_CLI_VERSION";
|
|
8
|
-
export declare const DEFAULT_GEMINI_CLI_VERSION = "0.
|
|
8
|
+
export declare const DEFAULT_GEMINI_CLI_VERSION = "0.52.0";
|
|
9
9
|
export declare function getGeminiCliUserAgent(modelId?: string): string;
|
|
10
10
|
export declare const getGeminiCliHeaders: (modelId?: string) => {
|
|
11
11
|
"User-Agent": string;
|
|
@@ -5,3 +5,7 @@ export interface GoogleVertexOptions extends GoogleSharedStreamOptions {
|
|
|
5
5
|
location?: string;
|
|
6
6
|
}
|
|
7
7
|
export declare const streamGoogleVertex: StreamFunction<"google-vertex">;
|
|
8
|
+
/** Test seam: the Vertex API key as resolved from options plus trusted env. */
|
|
9
|
+
export declare function resolveVertexApiKeyForTest(options?: GoogleVertexOptions): string | undefined;
|
|
10
|
+
/** Test seam: the Vertex location as resolved from options plus trusted env. */
|
|
11
|
+
export declare function resolveVertexLocationForTest(options?: GoogleVertexOptions): string;
|
|
@@ -10,6 +10,10 @@ export interface OpenAICodexResponsesOptions extends StreamOptions {
|
|
|
10
10
|
preferWebsockets?: boolean;
|
|
11
11
|
serviceTier?: ServiceTier;
|
|
12
12
|
}
|
|
13
|
+
/** Maps a canonical tool name to the name Codex accepts on the wire. */
|
|
14
|
+
export declare function codexToolWireName(name: string): string;
|
|
15
|
+
/** Maps a Codex wire tool name back to the canonical harness tool name. */
|
|
16
|
+
export declare function codexToolCanonicalName(wireName: string): string;
|
|
13
17
|
type CodexTransport = "sse" | "websocket";
|
|
14
18
|
export interface OpenAICodexWebSocketDebugStats {
|
|
15
19
|
fullContextRequests: number;
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
import type { ChatCompletionMessageParam } from "openai/resources/chat/completions";
|
|
2
2
|
import { type AssistantMessage, type Context, type Model, type ServiceTier, type StreamFunction, type StreamOptions, type ToolChoice } from "../types";
|
|
3
3
|
import { type ResolvedOpenAICompat } from "./openai-completions-compat";
|
|
4
|
+
/** Test seam: the provider base URL as resolved from trusted env. */
|
|
5
|
+
export declare function resolveOpenAICompletionsBaseUrlForTest(baseUrl: string | undefined, authCredentialType: "api_key" | "oauth" | undefined): string;
|
|
4
6
|
/**
|
|
5
7
|
* Identify "real progress" stream chunks vs. keepalives, role-only preambles,
|
|
6
8
|
* and empty `{choices:[]}` no-ops emitted by some OpenAI-compatible endpoints.
|
|
@@ -13,6 +13,8 @@ export interface OpenAIResponsesOptions extends StreamOptions {
|
|
|
13
13
|
*/
|
|
14
14
|
strictResponsesPairing?: boolean;
|
|
15
15
|
}
|
|
16
|
+
/** Test seam: the provider base URL as resolved from trusted env. */
|
|
17
|
+
export declare function resolveOpenAIProviderBaseUrlForTest(baseUrl: string | undefined, authCredentialType: "api_key" | "oauth" | undefined): string;
|
|
16
18
|
/**
|
|
17
19
|
* Generate function for OpenAI Responses API
|
|
18
20
|
*/
|
package/dist/types/types.d.ts
CHANGED
|
@@ -51,7 +51,7 @@ export interface ThinkingConfig {
|
|
|
51
51
|
/** Provider-specific transport used to encode the selected effort. */
|
|
52
52
|
mode: ThinkingControlMode;
|
|
53
53
|
}
|
|
54
|
-
export type KnownProvider = "alibaba-token-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "opengateway" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
|
|
54
|
+
export type KnownProvider = "alibaba-token-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "deepinfra" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "opengateway" | "bizrouter" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
|
|
55
55
|
export type Provider = KnownProvider | string;
|
|
56
56
|
import type { Effort } from "./model-thinking";
|
|
57
57
|
/** Token budgets for each thinking level (token-based providers only) */
|
|
@@ -1,10 +1,12 @@
|
|
|
1
|
-
import type { CredentialRankingStrategy, UsageProvider } from "../usage";
|
|
1
|
+
import type { CredentialRankingStrategy, UsageFetchParams, UsageProvider } from "../usage";
|
|
2
2
|
interface BillingUsage {
|
|
3
3
|
monthlyLimit: number;
|
|
4
4
|
used: number;
|
|
5
5
|
billingPeriodEnd: string;
|
|
6
6
|
}
|
|
7
7
|
export declare function parseGrokCliBillingUsage(payload: unknown): BillingUsage;
|
|
8
|
+
/** Test seam: the usage access token as resolved from a credential plus trusted env. */
|
|
9
|
+
export declare function resolveGrokAccessTokenForTest(params: UsageFetchParams): string | undefined;
|
|
8
10
|
export declare const grokCliUsageProvider: UsageProvider;
|
|
9
11
|
export declare const grokCliRankingStrategy: CredentialRankingStrategy;
|
|
10
12
|
export {};
|
|
@@ -1,2 +1,4 @@
|
|
|
1
1
|
import type { UsageProvider } from "../usage";
|
|
2
|
+
/** Test seam: the usage base URL as resolved from a caller value plus trusted env. */
|
|
3
|
+
export declare function normalizeKimiUsageBaseUrlForTest(baseUrl?: string): string;
|
|
2
4
|
export declare const kimiUsageProvider: UsageProvider;
|
|
@@ -4,6 +4,14 @@ export interface AnthropicAuthConfig {
|
|
|
4
4
|
baseUrl: string;
|
|
5
5
|
isOAuth: boolean;
|
|
6
6
|
}
|
|
7
|
+
/**
|
|
8
|
+
* Resolve the Anthropic base URL from the environment.
|
|
9
|
+
*
|
|
10
|
+
* Trusted sources only: the result becomes the request URL that carries the
|
|
11
|
+
* Anthropic API key / OAuth token, so whatever can set it can redirect
|
|
12
|
+
* authenticated traffic. `$env` merges the caller's `cwd/.env`, so reading it
|
|
13
|
+
* there would let repository content choose where credentials are sent.
|
|
14
|
+
*/
|
|
7
15
|
export declare function resolveAnthropicBaseUrlFromEnv(): string | undefined;
|
|
8
16
|
/**
|
|
9
17
|
* Checks if a token is an OAuth token by looking for sk-ant-oat prefix.
|
|
@@ -3,6 +3,8 @@ export interface FallbackTrigger {
|
|
|
3
3
|
class: FallbackTriggerClass;
|
|
4
4
|
retryAfterMs?: number;
|
|
5
5
|
}
|
|
6
|
+
/** Stable code for streams that time out before producing semantic progress. */
|
|
7
|
+
export declare const STREAM_FIRST_EVENT_TIMEOUT_PROVIDER_CODE = "stream_first_event_timeout";
|
|
6
8
|
export type TransportHeaders = Headers | Record<string, string | undefined>;
|
|
7
9
|
/**
|
|
8
10
|
* Structured facts from an upstream HTTP or transport failure. Retry decisions
|
|
@@ -1 +1,11 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Whether Anthropic requests run in Foundry gateway mode.
|
|
3
|
+
*
|
|
4
|
+
* Resolved from trusted environment sources only. Enabling Foundry switches the
|
|
5
|
+
* request base URL and injects TLS client material, so whatever can set this
|
|
6
|
+
* redirects authenticated traffic. `$env` merges the caller's `cwd/.env`, so
|
|
7
|
+
* reading it there would let repository content flip the mode; resolve it the
|
|
8
|
+
* same way the credentials themselves are (launching shell plus GJC/user-owned
|
|
9
|
+
* `.env` files, never the project `.env`).
|
|
10
|
+
*/
|
|
1
11
|
export declare function isFoundryEnabled(): boolean;
|
|
@@ -17,7 +17,30 @@ export type CapturedHttpErrorResponse = {
|
|
|
17
17
|
export declare function isModelUnavailableError(message: string, error: unknown): boolean;
|
|
18
18
|
/** Actionable guidance for selecting an available model/provider. */
|
|
19
19
|
export declare function formatModelUnavailableGuidance(dump: RawHttpRequestDump | undefined): string;
|
|
20
|
+
/** Directory holding the retained HTTP 400 dumps. */
|
|
21
|
+
export declare function httpRequestDumpDir(): string;
|
|
22
|
+
/**
|
|
23
|
+
* Drop the oldest dumps beyond the cap. Best-effort: diagnostics must never turn
|
|
24
|
+
* a request failure into a second failure, so every step swallows its error.
|
|
25
|
+
*
|
|
26
|
+
* File names are `${Date.now()}-${hash}.json`, so a lexical sort is chronological
|
|
27
|
+
* for the millisecond timestamps this writer produces.
|
|
28
|
+
*/
|
|
29
|
+
export declare function pruneHttpRequestDumps(dir?: string): Promise<number>;
|
|
20
30
|
export declare function appendRawHttpRequestDumpFor400(message: string, error: unknown, dump: RawHttpRequestDump | undefined): Promise<string>;
|
|
31
|
+
/**
|
|
32
|
+
* Name the failed connection when the request never produced an HTTP status.
|
|
33
|
+
*
|
|
34
|
+
* Bun raises DNS and socket failures as a bare `Error` whose message is a
|
|
35
|
+
* standalone hint ("Was there a typo in the url or port?", "Unable to connect.
|
|
36
|
+
* Is the computer able to access the url?") while the actionable facts live on
|
|
37
|
+
* `code` and `path`. Those properties are dropped when only `message` reaches
|
|
38
|
+
* the assistant message, so a provider outage, a local DNS failure, and a
|
|
39
|
+
* mistyped custom base URL all render as the same context-free sentence.
|
|
40
|
+
* Appending the code and the target URL tells the user which host failed and
|
|
41
|
+
* whether the fault is theirs.
|
|
42
|
+
*/
|
|
43
|
+
export declare function appendTransportFailureContext(message: string, error: unknown, rawRequestDump: RawHttpRequestDump | undefined): string;
|
|
21
44
|
export declare function finalizeErrorMessage(error: unknown, rawRequestDump: RawHttpRequestDump | undefined, capturedErrorResponse?: CapturedHttpErrorResponse): Promise<string>;
|
|
22
45
|
export declare function withHttpStatus(error: unknown, status: number): Error;
|
|
23
46
|
/**
|
|
@@ -30,6 +30,10 @@ export declare function getOpenAIStreamIdleTimeoutMs(): number | undefined;
|
|
|
30
30
|
*/
|
|
31
31
|
export declare function getStreamFirstEventTimeoutMs(idleTimeoutMs?: number, fallbackMs?: number): number | undefined;
|
|
32
32
|
export type Watchdog = NodeJS.Timeout | undefined;
|
|
33
|
+
export declare class FirstEventTimeoutError extends Error {
|
|
34
|
+
readonly providerCode = "stream_first_event_timeout";
|
|
35
|
+
constructor(message: string);
|
|
36
|
+
}
|
|
33
37
|
/**
|
|
34
38
|
* Starts a watchdog that aborts a request if no first stream event arrives in time.
|
|
35
39
|
* Call `markFirstEventReceived()` as soon as the first event is observed.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export declare const loginBizRouter: (options: import("./types").OAuthController) => Promise<string>;
|
|
@@ -2,6 +2,8 @@
|
|
|
2
2
|
* Kimi Code OAuth flow (device authorization grant)
|
|
3
3
|
*/
|
|
4
4
|
import type { OAuthController, OAuthCredentials } from "./types";
|
|
5
|
+
/** Test seam: the OAuth host as resolved from trusted env. */
|
|
6
|
+
export declare function resolveKimiOAuthHostForTest(): string;
|
|
5
7
|
export declare let getKimiCommonHeaders: () => Readonly<{
|
|
6
8
|
"User-Agent": `KimiCLI/${string}`;
|
|
7
9
|
"X-Msh-Platform": "kimi_cli";
|
|
@@ -1,9 +1,4 @@
|
|
|
1
1
|
import type { OAuthController, OAuthCredentials } from "./types";
|
|
2
|
-
/**
|
|
3
|
-
|
|
4
|
-
*
|
|
5
|
-
* Tries auto-extraction from the desktop app, then runs HTTP email OTP login.
|
|
6
|
-
*
|
|
7
|
-
* No browser/manual token paste fallback is used.
|
|
8
|
-
*/
|
|
2
|
+
/** Test seam: the resolved native-app borrowing opt-out. */
|
|
3
|
+
export declare function authBorrowDisabledForTest(): boolean;
|
|
9
4
|
export declare function loginPerplexity(ctrl: OAuthController): Promise<OAuthCredentials>;
|
|
@@ -7,7 +7,7 @@ export type OAuthCredentials = {
|
|
|
7
7
|
email?: string;
|
|
8
8
|
accountId?: string;
|
|
9
9
|
};
|
|
10
|
-
export type OAuthProvider = "alibaba-token-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "opengateway" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
|
|
10
|
+
export type OAuthProvider = "alibaba-token-plan" | "anthropic" | "bizrouter" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "deepinfra" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "opengateway" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
|
|
11
11
|
export type OAuthProviderId = OAuthProvider | (string & {});
|
|
12
12
|
export type OAuthPrompt = {
|
|
13
13
|
message: string;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@gajae-code/ai",
|
|
4
|
-
"version": "0.
|
|
4
|
+
"version": "0.12.1",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://gajae-code.com",
|
|
7
7
|
"author": "Yeachan-Heo and Gajae Code Contributors",
|
|
@@ -40,7 +40,7 @@
|
|
|
40
40
|
"dependencies": {
|
|
41
41
|
"@anthropic-ai/sdk": "^0.94.0",
|
|
42
42
|
"@bufbuild/protobuf": "^2.12.0",
|
|
43
|
-
"@gajae-code/utils": "0.
|
|
43
|
+
"@gajae-code/utils": "0.12.1",
|
|
44
44
|
"openai": "^6.36.0",
|
|
45
45
|
"partial-json": "^0.1.7",
|
|
46
46
|
"zod": "4.4.3"
|
package/src/auth-storage.ts
CHANGED
|
@@ -1998,6 +1998,12 @@ export class AuthStorage {
|
|
|
1998
1998
|
await saveApiKeyCredential(apiKey);
|
|
1999
1999
|
return;
|
|
2000
2000
|
}
|
|
2001
|
+
case "bizrouter": {
|
|
2002
|
+
const { loginBizRouter } = await import("./utils/oauth/bizrouter");
|
|
2003
|
+
const apiKey = await loginBizRouter(ctrl);
|
|
2004
|
+
await saveApiKeyCredential(apiKey);
|
|
2005
|
+
return;
|
|
2006
|
+
}
|
|
2001
2007
|
case "opengateway": {
|
|
2002
2008
|
const { loginOpenGateway } = await import("./utils/oauth/opengateway");
|
|
2003
2009
|
const apiKey = await loginOpenGateway(ctrl);
|
package/src/cli.ts
CHANGED
package/src/model-thinking.ts
CHANGED
|
@@ -350,6 +350,25 @@ export function hasOpus47ApiRestrictions(modelId: string): boolean {
|
|
|
350
350
|
return semverGte(parsed.version, "4.7") && parsed.kind === "opus";
|
|
351
351
|
}
|
|
352
352
|
|
|
353
|
+
/**
|
|
354
|
+
* Adaptive thinking `display` is supported starting with Anthropic Opus 4.7.
|
|
355
|
+
* Older adaptive-thinking models (Opus 4.6, Sonnet 4.6+) reject the field.
|
|
356
|
+
* Fable (5+) postdates Opus 4.7, accepts `display`, and defaults it to
|
|
357
|
+
* "omitted" — thinking tokens are billed but no content streams back — so it
|
|
358
|
+
* must opt in like Opus 4.7+ (issue #2791).
|
|
359
|
+
*
|
|
360
|
+
* Shares `hasOpus47ApiRestrictions` version parsing on purpose: the two
|
|
361
|
+
* predicates describe the same API generation, and a private `claude-opus-(\d+)-(\d+)`
|
|
362
|
+
* regex silently disagreed with it for single-component aliases (`claude-opus-5`
|
|
363
|
+
* matched nothing while `claude-opus-5-20260101` matched), so the same model sent
|
|
364
|
+
* a different thinking shape and beta set depending on which id string was used.
|
|
365
|
+
* Bedrock region/inference-profile prefixes are handled by the canonical parser.
|
|
366
|
+
*/
|
|
367
|
+
export function supportsAnthropicAdaptiveThinkingDisplay(modelId: string): boolean {
|
|
368
|
+
if (/claude-fable-\d/.test(modelId)) return true;
|
|
369
|
+
return hasOpus47ApiRestrictions(modelId);
|
|
370
|
+
}
|
|
371
|
+
|
|
353
372
|
function anthropicModelHasRealXHighEffort<TApi extends Api>(model: ApiModel<TApi>): boolean {
|
|
354
373
|
if (model.api !== "anthropic-messages") return false;
|
|
355
374
|
const parsedModel = parseKnownModel(model.id);
|
package/src/models.json
CHANGED
|
@@ -4030,6 +4030,78 @@
|
|
|
4030
4030
|
}
|
|
4031
4031
|
}
|
|
4032
4032
|
},
|
|
4033
|
+
"bizrouter": {
|
|
4034
|
+
"anthropic/claude-sonnet-4.5": {
|
|
4035
|
+
"id": "anthropic/claude-sonnet-4.5",
|
|
4036
|
+
"name": "Anthropic Sonnet 4.5",
|
|
4037
|
+
"api": "openai-completions",
|
|
4038
|
+
"provider": "bizrouter",
|
|
4039
|
+
"baseUrl": "https://api.bizrouter.ai/v1",
|
|
4040
|
+
"reasoning": true,
|
|
4041
|
+
"input": [
|
|
4042
|
+
"text",
|
|
4043
|
+
"image"
|
|
4044
|
+
],
|
|
4045
|
+
"cost": {
|
|
4046
|
+
"input": 3,
|
|
4047
|
+
"output": 15,
|
|
4048
|
+
"cacheRead": 0.3,
|
|
4049
|
+
"cacheWrite": 3.75
|
|
4050
|
+
},
|
|
4051
|
+
"contextWindow": 200000,
|
|
4052
|
+
"maxTokens": 64000,
|
|
4053
|
+
"thinking": {
|
|
4054
|
+
"mode": "effort",
|
|
4055
|
+
"minLevel": "minimal",
|
|
4056
|
+
"maxLevel": "xhigh"
|
|
4057
|
+
}
|
|
4058
|
+
},
|
|
4059
|
+
"google/gemini-2.5-pro": {
|
|
4060
|
+
"id": "google/gemini-2.5-pro",
|
|
4061
|
+
"name": "Gemini 2.5 Pro",
|
|
4062
|
+
"api": "openai-completions",
|
|
4063
|
+
"provider": "bizrouter",
|
|
4064
|
+
"baseUrl": "https://api.bizrouter.ai/v1",
|
|
4065
|
+
"reasoning": true,
|
|
4066
|
+
"input": [
|
|
4067
|
+
"text",
|
|
4068
|
+
"image"
|
|
4069
|
+
],
|
|
4070
|
+
"cost": {
|
|
4071
|
+
"input": 1.25,
|
|
4072
|
+
"output": 10,
|
|
4073
|
+
"cacheRead": 0.31,
|
|
4074
|
+
"cacheWrite": 0
|
|
4075
|
+
},
|
|
4076
|
+
"contextWindow": 1048576,
|
|
4077
|
+
"maxTokens": 65536,
|
|
4078
|
+
"thinking": {
|
|
4079
|
+
"mode": "effort",
|
|
4080
|
+
"minLevel": "minimal",
|
|
4081
|
+
"maxLevel": "high"
|
|
4082
|
+
}
|
|
4083
|
+
},
|
|
4084
|
+
"openai/gpt-4o": {
|
|
4085
|
+
"id": "openai/gpt-4o",
|
|
4086
|
+
"name": "GPT-4o",
|
|
4087
|
+
"api": "openai-completions",
|
|
4088
|
+
"provider": "bizrouter",
|
|
4089
|
+
"baseUrl": "https://api.bizrouter.ai/v1",
|
|
4090
|
+
"reasoning": false,
|
|
4091
|
+
"input": [
|
|
4092
|
+
"text",
|
|
4093
|
+
"image"
|
|
4094
|
+
],
|
|
4095
|
+
"cost": {
|
|
4096
|
+
"input": 2.5,
|
|
4097
|
+
"output": 10,
|
|
4098
|
+
"cacheRead": 1.25,
|
|
4099
|
+
"cacheWrite": 0
|
|
4100
|
+
},
|
|
4101
|
+
"contextWindow": 128000,
|
|
4102
|
+
"maxTokens": 16384
|
|
4103
|
+
}
|
|
4104
|
+
},
|
|
4033
4105
|
"cerebras": {
|
|
4034
4106
|
"gemma-4-31b": {
|
|
4035
4107
|
"id": "gemma-4-31b",
|
|
@@ -11,6 +11,7 @@ import { ollamaCloudModelManagerOptions } from "./ollama";
|
|
|
11
11
|
import {
|
|
12
12
|
alibabaTokenPlanModelManagerOptions,
|
|
13
13
|
anthropicModelManagerOptions,
|
|
14
|
+
bizrouterModelManagerOptions,
|
|
14
15
|
cerebrasModelManagerOptions,
|
|
15
16
|
cloudflareAiGatewayModelManagerOptions,
|
|
16
17
|
deepinfraModelManagerOptions,
|
|
@@ -319,6 +320,12 @@ export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [
|
|
|
319
320
|
config => opengatewayModelManagerOptions(config),
|
|
320
321
|
catalog("OpenGateway by Sionic AI", ["OPENGATEWAY_API_KEY"]),
|
|
321
322
|
),
|
|
323
|
+
catalogDescriptor(
|
|
324
|
+
"bizrouter",
|
|
325
|
+
"anthropic/claude-sonnet-4.5",
|
|
326
|
+
config => bizrouterModelManagerOptions(config),
|
|
327
|
+
catalog("BizRouter", ["BIZROUTER_API_KEY"]),
|
|
328
|
+
),
|
|
322
329
|
catalogDescriptor("zai", "glm-5.2", config => zaiModelManagerOptions(config), catalog("zAI", ["ZAI_API_KEY"])),
|
|
323
330
|
catalogDescriptor(
|
|
324
331
|
"glm-zcode",
|