@gajae-code/ai 0.12.11 → 0.12.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +27 -0
- package/dist/types/providers/anthropic.d.ts +13 -0
- package/dist/types/providers/openai-responses-shared.d.ts +1 -0
- package/dist/types/types.d.ts +5 -3
- package/dist/types/utils/idle-iterator.d.ts +17 -0
- package/dist/types/utils/tool-choice-capability.d.ts +5 -0
- package/dist/types/utils.d.ts +28 -0
- package/package.json +2 -2
- package/src/model-thinking.ts +34 -11
- package/src/models.json +52 -118
- package/src/provider-models/descriptors.ts +3 -3
- package/src/providers/anthropic.ts +57 -9
- package/src/providers/azure-openai-responses.ts +20 -4
- package/src/providers/openai-codex-responses.ts +31 -6
- package/src/providers/openai-completions.ts +3 -21
- package/src/providers/openai-responses-shared.ts +26 -0
- package/src/providers/openai-responses.ts +8 -26
- package/src/providers/register-builtins.ts +52 -27
- package/src/types.ts +9 -3
- package/src/utils/idle-iterator.ts +31 -0
- package/src/utils/tool-choice-capability.ts +33 -1
- package/src/utils/validation.ts +5 -0
- package/src/utils.ts +56 -0
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,32 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [0.12.13] - 2026-08-06
|
|
6
|
+
|
|
7
|
+
### Changed
|
|
8
|
+
|
|
9
|
+
- Anthropic prompt caching now defaults to top-level automatic caching (`cache_control: { type: "ephemeral" }`) for every Claude-family model, including through non-canonical Anthropic-compatible gateways (Cloudflare AI Gateway, GitHub Copilot, GitLab Duo, Vercel AI Gateway, zenmux, etc.), instead of only `api.anthropic.com`. Non-Claude models on unknown compatible endpoints keep the previous no-cache default; `compat.promptCacheMode: "none"`, `compat.promptCacheMode: "explicit"`, and configured or per-request `cacheRetention: "none"` still opt out. Non-canonical Claude models get the default ~5m cache lifetime unless the endpoint sets `compat.supportsLongCacheRetention: true`.
|
|
10
|
+
|
|
11
|
+
### Fixed
|
|
12
|
+
|
|
13
|
+
- `todo_write` raw argument rejections now carry bounded, authority-controlled correction codes for each rejected shape: unknown root keys, unknown operation-entry keys, done/drop entries missing a task or phase target, and unknown init list-entry keys. Each code maps to a fixed correction message naming the accepted shape (never echoing the offending input), so invalid calls surface specific guidance while valid payloads keep the existing passthrough/coercion path (#3916).
|
|
14
|
+
- Anthropic Sonnet 5 now exposes Anthropic's real `xhigh` and `max` thinking efforts on the Messages API (`minimal`/`low`/`medium`/`high`/`xhigh`/`max`), matching official support. The previous generic `kind === opus` gate excluded it from the full preset range; the capability predicate is now an explicit version-scoped list (Opus 4.7+, Sonnet 5+), so older Sonnet generations and Bedrock Converse routes stay fail-closed at their previously advertised levels (issue #3913).
|
|
15
|
+
- Alibaba Token Plan now exposes Qwen 3.8 Max under the provider-supported `qwen3.8-max` wire id instead of the rejected `qwen-3.8-max` spelling; catalog regeneration canonicalizes a legacy discovered alias rather than retaining a broken duplicate (#3909).
|
|
16
|
+
- Canonicalized first-class MiniMax M3 catalog ids (issue #3896). The bundled catalog previously shipped stale lowercase `minimax-m3` duplicates (512K) next to the canonical `MiniMax-M3` (1M) on all four first-class MiniMax providers, plus a non-official `minimax-v3` entry under `minimax-code`. The lowercase `minimax-m3` entries and `minimax-v3` are removed; `MiniMax-M3` is the single canonical first-class id (the regen-safe 1M pin in `applyGeneratedModelPolicy` now keys on `MiniMax-M3` / `MiniMax-M3[1m]` instead of the removed lowercase id), `DEFAULT_MODEL_PER_PROVIDER` points at `MiniMax-M3`, and the official Anthropic Token Plan id `MiniMax-M3[1m]` is first-class on the `minimax` / `minimax-cn` Anthropic routes with 1M context semantics. Unrelated catalog providers keep their own `minimax-m3` contracts.
|
|
17
|
+
- Anthropic thinking-replay repair now also triggers when the mutation/signature `invalid_request_error` arrives as a statusless in-stream SSE `error` event (issue #3900). Proxies such as CLIProxyAPI forward the upstream 400 body over an HTTP 200 SSE stream, so the thrown error carries no HTTP status; the classifiers previously required `status === 400` and let the session loop on an unrecoverable replay rejection. Statusless errors still require the full `invalid_request_error` thinking wording, so unrelated transport failures never claim the one-shot repair.
|
|
18
|
+
- Anthropic thinking-replay repair now also recovers when a proxy masks the rejection entirely (issue #3900). Live CLIProxyAPI captures replace the upstream 400 body with a generic `{"type":"api_error","message":"An error occurred while processing the request."}` SSE event on an HTTP 200 response, which names no cause and matches no transient phrase, so the turn died on the first attempt. Such a masked rejection now takes the same one-shot latest-then-full-history repair, but only before the first token and only while the request actually replays signed `thinking`/`redacted_thinking` blocks; masked failures on requests without replayed thinking still surface immediately. The classifier is exported as `isAnthropicMaskedProxyRejection`.
|
|
19
|
+
|
|
20
|
+
- Anthropic cache-control resolution now falls back to `model.cacheRetention` at the provider boundary, preserving configured retention and request-over-model precedence through special dispatch wrappers such as GitLab Duo. A configured `cacheRetention: "none"` can no longer be dropped and replaced by the new automatic Claude-family cache marker.
|
|
21
|
+
- Anthropic explicit prompt caching now advances its conversation breakpoint during tool-use loops by marking the latest completed assistant tool-use turn while leaving the newest tool result uncached. Previously it kept refreshing only the original human message until another human turn arrived, pinning proxy cache reads to the static tools/system prefix throughout long agentic runs.
|
|
22
|
+
## [0.12.12] - 2026-08-05
|
|
23
|
+
|
|
24
|
+
### Fixed
|
|
25
|
+
|
|
26
|
+
- OpenAI Responses and Azure OpenAI Responses now map the first-event timeout into the SDK request/setup timeout the same way Completions does, so a never-resolving pre-headers fetch on a provider-owned lazy stream cannot wait the SDK's 10-minute default before any transport watchdog exists. Alibaba Responses honors an explicit shorter first-event override before headers; Azure/env-pinned setup timeouts normalize to the typed `stream_first_event_timeout` failure.
|
|
27
|
+
- OpenAI Codex cost estimates now treat an explicit response `service_tier` as authoritative, so a request for priority processing that the provider serves at the default tier is no longer charged the priority multiplier; the requested tier remains the fallback when the terminal response omits the field.
|
|
28
|
+
- Added shared `isReasoningContentReplayError` classifier and `stripUnusableReasoningItems` repair for the DeepSeek-family reasoning-content replay rejection ("The `reasoning_content` in the thinking mode must be passed back to the API"). The classifier detects the error across message carrier shapes; the repair removes only `reasoning` items whose `encrypted_content` a proxy stripped to empty, preserving all non-reasoning history (text, tool calls, tool outputs). The agent loop consumes both for a bounded repair-and-resend circuit breaker.
|
|
29
|
+
- Codex statusless HTTP 200 SSE `invalid_request_error` events retry once without a forced named function choice only when the exact rejected name is still present in the request's serialized tools, before any output is emitted (#3669).
|
|
30
|
+
|
|
5
31
|
## [0.12.11] - 2026-08-03
|
|
6
32
|
|
|
7
33
|
## [0.12.10] - 2026-08-03
|
|
@@ -29,6 +55,7 @@
|
|
|
29
55
|
### Fixed
|
|
30
56
|
|
|
31
57
|
- Closed the two remaining ingress holes behind bare `Request Blocked` failures on OpenAI codex models. (1) The chatgpt.com/backend-api pre-model gate rejects with an HTTP 400 bare-`detail` body (`{"detail": "Request blocked."}`) carrying no `error.*` envelope and no `code=invalid_prompt`, so `parseCodexError` surfaced an unexplained message, `isInvalidPromptError` and the codex non-retryable classification missed it, and the session-level `invalid_prompt` circuit breaker never attempted a repaired resend. `parseCodexError` now reads top-level `detail` (string or `{message}`) bodies and classifies a leading `Request blocked` message without an explicit provider code as `invalid_prompt`, surfacing `Request blocked (code=invalid_prompt)` so every existing invalid_prompt contract engages. (2) Outgoing tool definitions (descriptions and JSON-schema strings) bypassed every request-boundary sanitizer on both the OpenAI Responses and OpenAI-codex-responses transports, so a `<|channel|>`-quoting MCP/skill tool description poisoned every request on the session in a way no history repair could fix. Both `convertTools` paths now neutralize reserved control tokens across the whole tool payload via the shared idempotent zero-width-space insertion (ref openai/codex#35838).
|
|
58
|
+
- Lazy built-in streams no longer place a normalized-event watchdog in front of providers that already monitor raw transport progress. This prevents active Anthropic, Azure OpenAI, and OpenAI-family streams from being replaced by a blank `Provider stream stalled while waiting for the next event` error when transport-only events refresh the provider watchdog; providers without their own watchdog keep the shared lazy-stream protection.
|
|
32
59
|
|
|
33
60
|
### Fixed
|
|
34
61
|
|
|
@@ -50,6 +50,19 @@ export declare function isAnthropicThinkingBlockMutationError(error: unknown): b
|
|
|
50
50
|
* than only the latest one.
|
|
51
51
|
*/
|
|
52
52
|
export declare function isAnthropicThinkingSignatureInvalidError(error: unknown): boolean;
|
|
53
|
+
/**
|
|
54
|
+
* CLIProxyAPI replaces Anthropic's rejection body wholesale instead of forwarding
|
|
55
|
+
* it: the client only ever sees
|
|
56
|
+
* `{"type":"error","error":{"type":"api_error","message":"An error occurred while
|
|
57
|
+
* processing the request."}}`, delivered as an in-stream SSE `error` event on an
|
|
58
|
+
* HTTP 200 response, so neither the status nor the message survives. Captured CPA
|
|
59
|
+
* traces for that masked shape carry the thinking-integrity 400 upstream (issue
|
|
60
|
+
* #3900), and the generic body matches no transient phrase either, so the turn
|
|
61
|
+
* dies unrecoverably. Nothing in the payload names the cause; callers must pair
|
|
62
|
+
* this with a request that actually replays signed thinking blocks before
|
|
63
|
+
* treating it as a thinking-replay rejection.
|
|
64
|
+
*/
|
|
65
|
+
export declare function isAnthropicMaskedProxyRejection(error: unknown): boolean;
|
|
53
66
|
export declare const claudeCodeVersion = "2.1.219";
|
|
54
67
|
export declare const claudeCodeEntrypoint = "sdk-cli";
|
|
55
68
|
export declare const claudeToolPrefix: string;
|
|
@@ -2,6 +2,7 @@ import type OpenAI from "openai";
|
|
|
2
2
|
import type { ResponseInput, ResponseInputContent, ResponseOutputItem } from "openai/resources/responses/responses";
|
|
3
3
|
import { type Api, type AssistantMessage, type ImageContent, type Model, type ServiceTier, type StopReason, type StreamOptions, type TextContent, type TextSignatureV1, type ToolCall, type ToolResultMessage } from "../types";
|
|
4
4
|
import type { AssistantMessageEventStream } from "../utils/event-stream";
|
|
5
|
+
export declare function isOpenAIResponsesProgressEvent(event: unknown): boolean;
|
|
5
6
|
export declare function encodeTextSignatureV1(id: string, phase?: TextSignatureV1["phase"]): string;
|
|
6
7
|
export declare function parseTextSignature(signature: string | undefined): {
|
|
7
8
|
id: string;
|
package/dist/types/types.d.ts
CHANGED
|
@@ -537,7 +537,7 @@ export type TSchema = ZodType | TJsonSchema;
|
|
|
537
537
|
export type Static<S> = S extends ZodType ? z.infer<S> : S extends {
|
|
538
538
|
static: infer T;
|
|
539
539
|
} ? T : unknown;
|
|
540
|
-
export type RawArgumentRejectionCode = "ask-intent-review-requires-positive-round" | "ask-intent-contract-requires-non-empty-authority" | "ask-deep-interview-metadata-requires-deep-interview-gate";
|
|
540
|
+
export type RawArgumentRejectionCode = "ask-intent-review-requires-positive-round" | "ask-intent-contract-requires-non-empty-authority" | "ask-deep-interview-metadata-requires-deep-interview-gate" | "todo-write-unknown-root-key" | "todo-write-unknown-op-entry-key" | "todo-write-done-drop-requires-target" | "todo-write-unknown-init-entry-key";
|
|
541
541
|
export type RawArgumentValidationResult = {
|
|
542
542
|
outcome: "passthrough";
|
|
543
543
|
} | {
|
|
@@ -790,8 +790,10 @@ export interface AnthropicCompat extends ToolChoiceCompat {
|
|
|
790
790
|
supportsLongCacheRetention?: boolean;
|
|
791
791
|
/**
|
|
792
792
|
* Prompt-cache transport accepted by this Anthropic-compatible endpoint.
|
|
793
|
-
* Canonical Anthropic
|
|
794
|
-
* to `"none"
|
|
793
|
+
* Canonical Anthropic and Claude-family models default to `"automatic"`;
|
|
794
|
+
* noncanonical non-Claude endpoints default to `"none"`. Set `"automatic"` to
|
|
795
|
+
* opt an otherwise unknown compatible endpoint into top-level caching, `"none"`
|
|
796
|
+
* to opt out, or `"explicit"` for endpoints that require block-level markers.
|
|
795
797
|
*/
|
|
796
798
|
promptCacheMode?: "none" | "explicit" | "automatic";
|
|
797
799
|
}
|
|
@@ -29,6 +29,23 @@ export declare function getOpenAIStreamIdleTimeoutMs(): number | undefined;
|
|
|
29
29
|
* env overrides still trump the fallback.
|
|
30
30
|
*/
|
|
31
31
|
export declare function getStreamFirstEventTimeoutMs(idleTimeoutMs?: number, fallbackMs?: number): number | undefined;
|
|
32
|
+
/**
|
|
33
|
+
* Resolves the OpenAI SDK client `timeout` so stalled-before-headers requests are
|
|
34
|
+
* bounded by the same first-event window the transport watchdog uses after
|
|
35
|
+
* `create()` returns. Without this, providers that only arm
|
|
36
|
+
* `iterateWithIdleTimeout` post-setup can wait the full SDK default (10 minutes
|
|
37
|
+
* per attempt) before any provider-owned watchdog exists.
|
|
38
|
+
*
|
|
39
|
+
* - Explicit `0` disables the request timeout (the SDK treats `timeout: 0` as an
|
|
40
|
+
* immediate failure, so callers that disable the first-event watchdog must not
|
|
41
|
+
* pass a timeout).
|
|
42
|
+
* - Providers with a first-event fallback (Alibaba, Kimi) honor an explicit
|
|
43
|
+
* nonzero override as-is, even when shorter than the fallback.
|
|
44
|
+
* - Other providers floor an explicit override at the env/default first-event
|
|
45
|
+
* window so a short post-connect first-event budget cannot kill legitimate
|
|
46
|
+
* slow setup.
|
|
47
|
+
*/
|
|
48
|
+
export declare function resolveOpenAISdkRequestTimeoutMs(provider: string, streamFirstEventTimeoutOverride?: number): number | undefined;
|
|
32
49
|
export type Watchdog = NodeJS.Timeout | undefined;
|
|
33
50
|
export declare class FirstEventTimeoutError extends Error {
|
|
34
51
|
readonly providerCode = "stream_first_event_timeout";
|
|
@@ -26,6 +26,11 @@ export declare function markToolChoiceIncapability(model: Model<Api>, maxSupport
|
|
|
26
26
|
export declare function resolveToolChoice(model: Model<Api>, requested: ToolChoice | undefined, compat?: ToolChoiceCompat): ResolveToolChoiceResult;
|
|
27
27
|
/** Detects provider errors indicating forced tool_choice is unsupported. */
|
|
28
28
|
export declare function isForcedToolChoiceUnsupportedError(error: unknown, sentForcedToolChoice: boolean): boolean;
|
|
29
|
+
/**
|
|
30
|
+
* Detects Codex's statusless SSE rejection for a named function tool choice.
|
|
31
|
+
* This is intentionally separate from the shared HTTP-400 classifier.
|
|
32
|
+
*/
|
|
33
|
+
export declare function isCodexStatuslessNamedToolChoiceNotFoundError(error: unknown, forcedToolName: string | undefined, sentToolNames: readonly string[]): boolean;
|
|
29
34
|
export type { ToolChoiceCompat, ToolChoiceSupport, ToolChoiceSupportSource } from "../types";
|
|
30
35
|
export interface ResolveToolChoiceResult {
|
|
31
36
|
requestedChoice: ToolChoice | undefined;
|
package/dist/types/utils.d.ts
CHANGED
|
@@ -59,6 +59,34 @@ export declare function neutralizeReservedControlTokens(text: string): string;
|
|
|
59
59
|
* uniformly testable across transports.
|
|
60
60
|
*/
|
|
61
61
|
export declare function isInvalidPromptError(input: unknown): boolean;
|
|
62
|
+
/**
|
|
63
|
+
* Shape-tolerant classifier for the DeepSeek-family reasoning-content replay
|
|
64
|
+
* rejection: "The `reasoning_content` in the thinking mode must be passed back
|
|
65
|
+
* to the API." DeepSeek V4 (and reasoning-capable siblings reached through any
|
|
66
|
+
* OpenAI-compatible proxy) 400 every follow-up turn once a prior assistant turn
|
|
67
|
+
* carried reasoning the proxy stripped to an empty `encrypted_content` /
|
|
68
|
+
* `reasoning_content`. Resending the identical history re-triggers it, so naive
|
|
69
|
+
* session auto-retry just burns the budget — it needs the same bounded
|
|
70
|
+
* repair-and-resend contract as the `invalid_prompt` poisoned-history breaker.
|
|
71
|
+
*
|
|
72
|
+
* Accepts a raw provider error, an assistant message, or any object carrying an
|
|
73
|
+
* `errorMessage` field (the agent-loop circuit breaker keys on this).
|
|
74
|
+
*/
|
|
75
|
+
export declare function isReasoningContentReplayError(input: unknown): boolean;
|
|
76
|
+
/**
|
|
77
|
+
* Remove Responses-API `reasoning` items whose `encrypted_content` is missing or
|
|
78
|
+
* empty from an outgoing history payload. DeepSeek rejects replay of reasoning
|
|
79
|
+
* whose encrypted blob a proxy stripped to `""`; dropping those items lets the
|
|
80
|
+
* model re-reason on the next turn instead of re-triggering a deterministic 400.
|
|
81
|
+
* Non-reasoning items (text, function_call, function_call_output, ...) are kept
|
|
82
|
+
* verbatim so tool-use pairing and message order are preserved. Returns whether
|
|
83
|
+
* any item was actually removed — the circuit breaker uses this to decide
|
|
84
|
+
* between a single repaired resend (removed) and immediate fail-fast (unchanged).
|
|
85
|
+
*/
|
|
86
|
+
export declare function stripUnusableReasoningItems(items: Array<Record<string, unknown>>): {
|
|
87
|
+
result: Array<Record<string, unknown>>;
|
|
88
|
+
removed: number;
|
|
89
|
+
};
|
|
62
90
|
/**
|
|
63
91
|
* Neutralize leaked reserved control tokens across every string in an outgoing
|
|
64
92
|
* Responses `input` array. This is the request-boundary complement to the
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@gajae-code/ai",
|
|
4
|
-
"version": "0.12.
|
|
4
|
+
"version": "0.12.13",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://gajae-code.com",
|
|
7
7
|
"author": "Yeachan-Heo and Gajae Code Contributors",
|
|
@@ -40,7 +40,7 @@
|
|
|
40
40
|
"dependencies": {
|
|
41
41
|
"@anthropic-ai/sdk": "^0.94.0",
|
|
42
42
|
"@bufbuild/protobuf": "^2.12.0",
|
|
43
|
-
"@gajae-code/utils": "0.12.
|
|
43
|
+
"@gajae-code/utils": "0.12.13",
|
|
44
44
|
"openai": "^6.36.0",
|
|
45
45
|
"partial-json": "^0.1.7",
|
|
46
46
|
"zod": "4.4.3"
|
package/src/model-thinking.ts
CHANGED
|
@@ -379,8 +379,18 @@ export function supportsAnthropicAdaptiveThinkingDisplay(modelId: string): boole
|
|
|
379
379
|
function anthropicModelHasRealXHighEffort<TApi extends Api>(model: ApiModel<TApi>): boolean {
|
|
380
380
|
if (model.api !== "anthropic-messages") return false;
|
|
381
381
|
const parsedModel = parseKnownModel(model.id);
|
|
382
|
-
if (parsedModel.family !== "anthropic"
|
|
383
|
-
|
|
382
|
+
if (parsedModel.family !== "anthropic") return false;
|
|
383
|
+
// Explicit capability predicate instead of a generic `kind === opus` gate:
|
|
384
|
+
// Sonnet 5 officially exposes Anthropic's real xhigh and max presets on
|
|
385
|
+
// the Messages API just like Opus 4.7+. Older Sonnet generations do not,
|
|
386
|
+
// so the predicate stays fail-closed for them.
|
|
387
|
+
if (parsedModel.kind === "opus") {
|
|
388
|
+
return semverGte(parsedModel.version, "4.7");
|
|
389
|
+
}
|
|
390
|
+
if (parsedModel.kind === "sonnet") {
|
|
391
|
+
return semverGte(parsedModel.version, "5.0");
|
|
392
|
+
}
|
|
393
|
+
return false;
|
|
384
394
|
}
|
|
385
395
|
|
|
386
396
|
function applyGeneratedModelPolicy(model: ApiModel<Api>): void {
|
|
@@ -456,11 +466,18 @@ function applyGeneratedModelPolicy(model: ApiModel<Api>): void {
|
|
|
456
466
|
requiresReasoningContentForToolCalls: true,
|
|
457
467
|
};
|
|
458
468
|
}
|
|
459
|
-
// MiniMax-M3
|
|
460
|
-
//
|
|
461
|
-
//
|
|
462
|
-
|
|
463
|
-
|
|
469
|
+
// MiniMax-M3's official Token Plan routes expose a 1M context window.
|
|
470
|
+
// Scope the correction to the four first-class regional MiniMax routes
|
|
471
|
+
// (canonical id plus the Anthropic Token Plan `[1m]` id); unrelated
|
|
472
|
+
// catalog aliases and providers keep their own contracts.
|
|
473
|
+
if (
|
|
474
|
+
(model.id === "MiniMax-M3" || model.id === "MiniMax-M3[1m]") &&
|
|
475
|
+
(model.provider === "minimax" ||
|
|
476
|
+
model.provider === "minimax-cn" ||
|
|
477
|
+
model.provider === "minimax-code" ||
|
|
478
|
+
model.provider === "minimax-code-cn")
|
|
479
|
+
) {
|
|
480
|
+
model.contextWindow = 1_000_000;
|
|
464
481
|
}
|
|
465
482
|
}
|
|
466
483
|
|
|
@@ -684,10 +701,16 @@ function inferAnthropicSupportedEfforts<TApi extends Api>(
|
|
|
684
701
|
// Converse lacks it (same split as Opus 4.7+ below).
|
|
685
702
|
return model.api === "anthropic-messages" ? DEFAULT_REASONING_EFFORTS_WITH_XHIGH : DEFAULT_REASONING_EFFORTS;
|
|
686
703
|
}
|
|
687
|
-
if (
|
|
688
|
-
|
|
689
|
-
|
|
690
|
-
|
|
704
|
+
if (anthropicModelHasRealXHighEffort(model)) {
|
|
705
|
+
// Opus 4.7+ and Sonnet 5 expose both Anthropic's real xhigh and
|
|
706
|
+
// max presets on the Messages API.
|
|
707
|
+
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH_AND_MAX;
|
|
708
|
+
}
|
|
709
|
+
if (parsedModel.kind === "opus") {
|
|
710
|
+
// Opus 4.6 exposes max but not the newer xhigh literal.
|
|
711
|
+
return DEFAULT_REASONING_EFFORTS_WITH_MAX;
|
|
712
|
+
}
|
|
713
|
+
return DEFAULT_REASONING_EFFORTS;
|
|
691
714
|
}
|
|
692
715
|
return inferFallbackEfforts(model);
|
|
693
716
|
}
|
package/src/models.json
CHANGED
|
@@ -89,6 +89,33 @@
|
|
|
89
89
|
"maxLevel": "xhigh"
|
|
90
90
|
}
|
|
91
91
|
},
|
|
92
|
+
"qwen3.8-max": {
|
|
93
|
+
"id": "qwen3.8-max",
|
|
94
|
+
"name": "Qwen3.8 Max",
|
|
95
|
+
"api": "openai-responses",
|
|
96
|
+
"provider": "alibaba-token-plan",
|
|
97
|
+
"baseUrl": "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1",
|
|
98
|
+
"reasoning": true,
|
|
99
|
+
"input": [
|
|
100
|
+
"text"
|
|
101
|
+
],
|
|
102
|
+
"cost": {
|
|
103
|
+
"input": 0,
|
|
104
|
+
"output": 0,
|
|
105
|
+
"cacheRead": 0,
|
|
106
|
+
"cacheWrite": 0
|
|
107
|
+
},
|
|
108
|
+
"contextWindow": 1000000,
|
|
109
|
+
"maxTokens": 65536,
|
|
110
|
+
"compat": {
|
|
111
|
+
"supportsDeveloperRole": false
|
|
112
|
+
},
|
|
113
|
+
"thinking": {
|
|
114
|
+
"mode": "effort",
|
|
115
|
+
"minLevel": "minimal",
|
|
116
|
+
"maxLevel": "xhigh"
|
|
117
|
+
}
|
|
118
|
+
},
|
|
92
119
|
"qwen3.8-max-preview": {
|
|
93
120
|
"id": "qwen3.8-max-preview",
|
|
94
121
|
"name": "Qwen3.8 Max Preview",
|
|
@@ -3925,7 +3952,7 @@
|
|
|
3925
3952
|
"thinking": {
|
|
3926
3953
|
"mode": "anthropic-adaptive",
|
|
3927
3954
|
"minLevel": "minimal",
|
|
3928
|
-
"maxLevel": "
|
|
3955
|
+
"maxLevel": "max"
|
|
3929
3956
|
}
|
|
3930
3957
|
},
|
|
3931
3958
|
"claude-opus-5": {
|
|
@@ -4713,7 +4740,7 @@
|
|
|
4713
4740
|
"thinking": {
|
|
4714
4741
|
"mode": "anthropic-adaptive",
|
|
4715
4742
|
"minLevel": "minimal",
|
|
4716
|
-
"maxLevel": "
|
|
4743
|
+
"maxLevel": "max"
|
|
4717
4744
|
}
|
|
4718
4745
|
},
|
|
4719
4746
|
"claude-sonnet-4-5": {
|
|
@@ -9839,7 +9866,7 @@
|
|
|
9839
9866
|
"thinking": {
|
|
9840
9867
|
"mode": "anthropic-adaptive",
|
|
9841
9868
|
"minLevel": "minimal",
|
|
9842
|
-
"maxLevel": "
|
|
9869
|
+
"maxLevel": "max"
|
|
9843
9870
|
}
|
|
9844
9871
|
},
|
|
9845
9872
|
"gemini-2.5-pro": {
|
|
@@ -40174,8 +40201,8 @@
|
|
|
40174
40201
|
"maxLevel": "xhigh"
|
|
40175
40202
|
}
|
|
40176
40203
|
},
|
|
40177
|
-
"
|
|
40178
|
-
"id": "
|
|
40204
|
+
"MiniMax-M3": {
|
|
40205
|
+
"id": "MiniMax-M3",
|
|
40179
40206
|
"name": "MiniMax-M3",
|
|
40180
40207
|
"api": "anthropic-messages",
|
|
40181
40208
|
"provider": "minimax",
|
|
@@ -40186,12 +40213,12 @@
|
|
|
40186
40213
|
"image"
|
|
40187
40214
|
],
|
|
40188
40215
|
"cost": {
|
|
40189
|
-
"input": 0.
|
|
40190
|
-
"output": 2
|
|
40191
|
-
"cacheRead": 0.
|
|
40216
|
+
"input": 0.3,
|
|
40217
|
+
"output": 1.2,
|
|
40218
|
+
"cacheRead": 0.06,
|
|
40192
40219
|
"cacheWrite": 0
|
|
40193
40220
|
},
|
|
40194
|
-
"contextWindow":
|
|
40221
|
+
"contextWindow": 1000000,
|
|
40195
40222
|
"maxTokens": 128000,
|
|
40196
40223
|
"thinking": {
|
|
40197
40224
|
"mode": "budget",
|
|
@@ -40199,9 +40226,9 @@
|
|
|
40199
40226
|
"maxLevel": "xhigh"
|
|
40200
40227
|
}
|
|
40201
40228
|
},
|
|
40202
|
-
"MiniMax-M3": {
|
|
40203
|
-
"id": "MiniMax-M3",
|
|
40204
|
-
"name": "MiniMax-M3",
|
|
40229
|
+
"MiniMax-M3[1m]": {
|
|
40230
|
+
"id": "MiniMax-M3[1m]",
|
|
40231
|
+
"name": "MiniMax-M3[1m]",
|
|
40205
40232
|
"api": "anthropic-messages",
|
|
40206
40233
|
"provider": "minimax",
|
|
40207
40234
|
"baseUrl": "https://api.minimax.io/anthropic",
|
|
@@ -40394,8 +40421,8 @@
|
|
|
40394
40421
|
"maxLevel": "xhigh"
|
|
40395
40422
|
}
|
|
40396
40423
|
},
|
|
40397
|
-
"
|
|
40398
|
-
"id": "
|
|
40424
|
+
"MiniMax-M3": {
|
|
40425
|
+
"id": "MiniMax-M3",
|
|
40399
40426
|
"name": "MiniMax-M3",
|
|
40400
40427
|
"api": "anthropic-messages",
|
|
40401
40428
|
"provider": "minimax-cn",
|
|
@@ -40406,12 +40433,12 @@
|
|
|
40406
40433
|
"image"
|
|
40407
40434
|
],
|
|
40408
40435
|
"cost": {
|
|
40409
|
-
"input": 0.
|
|
40410
|
-
"output": 2
|
|
40411
|
-
"cacheRead": 0.
|
|
40436
|
+
"input": 0.3,
|
|
40437
|
+
"output": 1.2,
|
|
40438
|
+
"cacheRead": 0.06,
|
|
40412
40439
|
"cacheWrite": 0
|
|
40413
40440
|
},
|
|
40414
|
-
"contextWindow":
|
|
40441
|
+
"contextWindow": 1000000,
|
|
40415
40442
|
"maxTokens": 128000,
|
|
40416
40443
|
"thinking": {
|
|
40417
40444
|
"mode": "budget",
|
|
@@ -40419,9 +40446,9 @@
|
|
|
40419
40446
|
"maxLevel": "xhigh"
|
|
40420
40447
|
}
|
|
40421
40448
|
},
|
|
40422
|
-
"MiniMax-M3": {
|
|
40423
|
-
"id": "MiniMax-M3",
|
|
40424
|
-
"name": "MiniMax-M3",
|
|
40449
|
+
"MiniMax-M3[1m]": {
|
|
40450
|
+
"id": "MiniMax-M3[1m]",
|
|
40451
|
+
"name": "MiniMax-M3[1m]",
|
|
40425
40452
|
"api": "anthropic-messages",
|
|
40426
40453
|
"provider": "minimax-cn",
|
|
40427
40454
|
"baseUrl": "https://api.minimaxi.com/anthropic",
|
|
@@ -40686,37 +40713,6 @@
|
|
|
40686
40713
|
"maxLevel": "high"
|
|
40687
40714
|
}
|
|
40688
40715
|
},
|
|
40689
|
-
"minimax-m3": {
|
|
40690
|
-
"id": "minimax-m3",
|
|
40691
|
-
"name": "MiniMax-M3",
|
|
40692
|
-
"api": "openai-completions",
|
|
40693
|
-
"provider": "minimax-code",
|
|
40694
|
-
"baseUrl": "https://api.minimax.io/v1",
|
|
40695
|
-
"reasoning": true,
|
|
40696
|
-
"input": [
|
|
40697
|
-
"text",
|
|
40698
|
-
"image"
|
|
40699
|
-
],
|
|
40700
|
-
"cost": {
|
|
40701
|
-
"input": 0,
|
|
40702
|
-
"output": 0,
|
|
40703
|
-
"cacheRead": 0,
|
|
40704
|
-
"cacheWrite": 0
|
|
40705
|
-
},
|
|
40706
|
-
"contextWindow": 512000,
|
|
40707
|
-
"maxTokens": 128000,
|
|
40708
|
-
"compat": {
|
|
40709
|
-
"supportsStore": false,
|
|
40710
|
-
"supportsDeveloperRole": false,
|
|
40711
|
-
"supportsReasoningEffort": false,
|
|
40712
|
-
"reasoningContentField": "reasoning_content"
|
|
40713
|
-
},
|
|
40714
|
-
"thinking": {
|
|
40715
|
-
"mode": "effort",
|
|
40716
|
-
"minLevel": "minimal",
|
|
40717
|
-
"maxLevel": "high"
|
|
40718
|
-
}
|
|
40719
|
-
},
|
|
40720
40716
|
"MiniMax-M3": {
|
|
40721
40717
|
"id": "MiniMax-M3",
|
|
40722
40718
|
"name": "MiniMax-M3",
|
|
@@ -40747,37 +40743,6 @@
|
|
|
40747
40743
|
"minLevel": "minimal",
|
|
40748
40744
|
"maxLevel": "high"
|
|
40749
40745
|
}
|
|
40750
|
-
},
|
|
40751
|
-
"minimax-v3": {
|
|
40752
|
-
"id": "minimax-v3",
|
|
40753
|
-
"name": "MiniMax-V3",
|
|
40754
|
-
"api": "openai-completions",
|
|
40755
|
-
"provider": "minimax-code",
|
|
40756
|
-
"baseUrl": "https://api.minimax.io/v1",
|
|
40757
|
-
"reasoning": true,
|
|
40758
|
-
"input": [
|
|
40759
|
-
"text",
|
|
40760
|
-
"image"
|
|
40761
|
-
],
|
|
40762
|
-
"cost": {
|
|
40763
|
-
"input": 0,
|
|
40764
|
-
"output": 0,
|
|
40765
|
-
"cacheRead": 0,
|
|
40766
|
-
"cacheWrite": 0
|
|
40767
|
-
},
|
|
40768
|
-
"contextWindow": 512000,
|
|
40769
|
-
"maxTokens": 128000,
|
|
40770
|
-
"compat": {
|
|
40771
|
-
"supportsStore": false,
|
|
40772
|
-
"supportsDeveloperRole": false,
|
|
40773
|
-
"supportsReasoningEffort": false,
|
|
40774
|
-
"reasoningContentField": "reasoning_content"
|
|
40775
|
-
},
|
|
40776
|
-
"thinking": {
|
|
40777
|
-
"mode": "effort",
|
|
40778
|
-
"minLevel": "minimal",
|
|
40779
|
-
"maxLevel": "high"
|
|
40780
|
-
}
|
|
40781
40746
|
}
|
|
40782
40747
|
},
|
|
40783
40748
|
"minimax-code-cn": {
|
|
@@ -41021,37 +40986,6 @@
|
|
|
41021
40986
|
"maxLevel": "high"
|
|
41022
40987
|
}
|
|
41023
40988
|
},
|
|
41024
|
-
"minimax-m3": {
|
|
41025
|
-
"id": "minimax-m3",
|
|
41026
|
-
"name": "MiniMax-M3",
|
|
41027
|
-
"api": "openai-completions",
|
|
41028
|
-
"provider": "minimax-code-cn",
|
|
41029
|
-
"baseUrl": "https://api.minimaxi.com/v1",
|
|
41030
|
-
"reasoning": true,
|
|
41031
|
-
"input": [
|
|
41032
|
-
"text",
|
|
41033
|
-
"image"
|
|
41034
|
-
],
|
|
41035
|
-
"cost": {
|
|
41036
|
-
"input": 0,
|
|
41037
|
-
"output": 0,
|
|
41038
|
-
"cacheRead": 0,
|
|
41039
|
-
"cacheWrite": 0
|
|
41040
|
-
},
|
|
41041
|
-
"contextWindow": 512000,
|
|
41042
|
-
"maxTokens": 128000,
|
|
41043
|
-
"compat": {
|
|
41044
|
-
"supportsStore": false,
|
|
41045
|
-
"supportsDeveloperRole": false,
|
|
41046
|
-
"supportsReasoningEffort": false,
|
|
41047
|
-
"reasoningContentField": "reasoning_content"
|
|
41048
|
-
},
|
|
41049
|
-
"thinking": {
|
|
41050
|
-
"mode": "effort",
|
|
41051
|
-
"minLevel": "minimal",
|
|
41052
|
-
"maxLevel": "high"
|
|
41053
|
-
}
|
|
41054
|
-
},
|
|
41055
40989
|
"MiniMax-M3": {
|
|
41056
40990
|
"id": "MiniMax-M3",
|
|
41057
40991
|
"name": "MiniMax-M3",
|
|
@@ -60418,7 +60352,7 @@
|
|
|
60418
60352
|
"thinking": {
|
|
60419
60353
|
"mode": "anthropic-adaptive",
|
|
60420
60354
|
"minLevel": "minimal",
|
|
60421
|
-
"maxLevel": "
|
|
60355
|
+
"maxLevel": "max"
|
|
60422
60356
|
}
|
|
60423
60357
|
},
|
|
60424
60358
|
"deepseek-v4-flash": {
|
|
@@ -75415,7 +75349,7 @@
|
|
|
75415
75349
|
"thinking": {
|
|
75416
75350
|
"mode": "anthropic-adaptive",
|
|
75417
75351
|
"minLevel": "minimal",
|
|
75418
|
-
"maxLevel": "
|
|
75352
|
+
"maxLevel": "max"
|
|
75419
75353
|
}
|
|
75420
75354
|
},
|
|
75421
75355
|
"arcee-ai/trinity-large-preview": {
|
|
@@ -81692,7 +81626,7 @@
|
|
|
81692
81626
|
"thinking": {
|
|
81693
81627
|
"mode": "anthropic-adaptive",
|
|
81694
81628
|
"minLevel": "minimal",
|
|
81695
|
-
"maxLevel": "
|
|
81629
|
+
"maxLevel": "max"
|
|
81696
81630
|
}
|
|
81697
81631
|
},
|
|
81698
81632
|
"anthropic/claude-sonnet-5-free": {
|
|
@@ -81717,7 +81651,7 @@
|
|
|
81717
81651
|
"thinking": {
|
|
81718
81652
|
"mode": "anthropic-adaptive",
|
|
81719
81653
|
"minLevel": "minimal",
|
|
81720
|
-
"maxLevel": "
|
|
81654
|
+
"maxLevel": "max"
|
|
81721
81655
|
}
|
|
81722
81656
|
},
|
|
81723
81657
|
"baidu/ernie-5.0-thinking-preview": {
|
|
@@ -365,9 +365,9 @@ export const DEFAULT_MODEL_PER_PROVIDER: Record<KnownProvider, string> = {
|
|
|
365
365
|
"google-antigravity": "gemini-3-pro-high",
|
|
366
366
|
"google-gemini-cli": "gemini-2.5-pro",
|
|
367
367
|
"google-vertex": "gemini-3-pro-preview",
|
|
368
|
-
minimax: "
|
|
369
|
-
"minimax-code": "
|
|
370
|
-
"minimax-code-cn": "
|
|
368
|
+
minimax: "MiniMax-M3",
|
|
369
|
+
"minimax-code": "MiniMax-M3",
|
|
370
|
+
"minimax-code-cn": "MiniMax-M3",
|
|
371
371
|
"openai-codex": "gpt-5.5",
|
|
372
372
|
"gitlab-duo": "duo-chat-sonnet-4-5",
|
|
373
373
|
} as Record<KnownProvider, string>;
|