@oh-my-pi/pi-catalog 17.4.2 → 18.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +23 -0
- package/dist/types/discovery/gemini-cli.d.ts +33 -0
- package/dist/types/discovery/index.d.ts +1 -0
- package/dist/types/hosts.d.ts +5 -0
- package/dist/types/identity/family.d.ts +1 -1
- package/dist/types/provider-models/google.d.ts +2 -0
- package/dist/types/types.d.ts +20 -2
- package/package.json +4 -4
- package/src/build.ts +2 -0
- package/src/compat/openai.ts +9 -4
- package/src/discovery/gemini-cli.ts +198 -0
- package/src/discovery/index.ts +1 -0
- package/src/discovery/protobuf.ts +3 -3
- package/src/hosts.ts +2 -0
- package/src/identity/family.ts +1 -1
- package/src/model-thinking.ts +42 -3
- package/src/models.json +11174 -3615
- package/src/provider-models/cache-provider-id.ts +3 -3
- package/src/provider-models/google.ts +10 -2
- package/src/provider-models/openai-compat.ts +50 -32
- package/src/types.ts +20 -1
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,29 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [18.0.1] - 2026-08-23
|
|
6
|
+
|
|
7
|
+
### Added
|
|
8
|
+
|
|
9
|
+
- Fixed `google-gemini-cli` model refresh returning only bundled models for Gemini Code Assist Standard accounts, whose credential is not authorized for the Antigravity `fetchAvailableModels` endpoint (HTTP 403). Discovery now falls back to the account's own `retrieveUserQuota` list on Cloud Code Assist, surfacing models such as `gemini-3.5-flash` ([#9315](https://github.com/can1357/oh-my-pi/issues/9315)).
|
|
10
|
+
|
|
11
|
+
### Fixed
|
|
12
|
+
|
|
13
|
+
- Fixed `opencode-go/ox-alpha-free` sending `reasoning_effort: "xhigh"` for the top thinking tier, which the OpenCode Go gateway rejects; the model now uses the gateway's wire-exact `low`/`high`/`max` ladder with mandatory thinking so `--thinking max` reaches the real max tier ([#9349](https://github.com/can1357/oh-my-pi/issues/9349)).
|
|
14
|
+
- Fixed Venice-hosted Qwen models (e.g. `venice/qwen3-6-35b-a3b`) failing with `400 Invalid request parameters`. Reasoning levels now use the accepted OpenAI-style `reasoning_effort` field, while Thinking Off sends Venice's explicit `venice_parameters.disable_thinking` flag ([#9345](https://github.com/can1357/oh-my-pi/issues/9345)).
|
|
15
|
+
- Fixed gateway-first OpenCode Zen and Go models missing context, output, image, and reasoning metadata by enriching live discovery from the current stencil catalog ([#9272](https://github.com/can1357/oh-my-pi/issues/9272)).
|
|
16
|
+
- Fixed `opencode-go/deepseek-v4-flash` exposing the generic `minimal`/`low`/`medium`/`high`/`xhigh` thinking ladder instead of DeepSeek V4's real `low`/`high`/`max` tiers. The model is pinned to the Responses transport (the Go gateway serves it only at `/responses`), which the DeepSeek effort branch did not admit, so it fell through to the default ladder; the branch now covers the `openai-responses` transport like every other host ([#9134](https://github.com/can1357/oh-my-pi/issues/9134)).
|
|
17
|
+
- Fixed protobuf map decoding corrupting entries when a key is `__proto__`, which dropped that argument and replayed spurious numeric arguments ([#9394](https://github.com/can1357/oh-my-pi/issues/9394)).
|
|
18
|
+
|
|
19
|
+
### Added
|
|
20
|
+
|
|
21
|
+
- Added Amazon Bedrock guardrail metadata to model definitions for Converse requests.
|
|
22
|
+
## [18.0.0] - 2026-08-22
|
|
23
|
+
|
|
24
|
+
### Added
|
|
25
|
+
|
|
26
|
+
- Added model capability metadata for reversible private-use glyph tokenization on Claude-compatible models, so provider request handling can apply the compatibility layer without inferring from transport details.
|
|
27
|
+
|
|
5
28
|
## [17.4.2] - 2026-08-21
|
|
6
29
|
|
|
7
30
|
### Added
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
import type { ModelSpec } from "../types.js";
|
|
2
|
+
import { type VariantCollapseTable } from "../variant-collapse.js";
|
|
3
|
+
/**
|
|
4
|
+
* Options for the Gemini CLI quota-based discovery fallback.
|
|
5
|
+
*/
|
|
6
|
+
export interface FetchGeminiCliQuotaModelsOptions {
|
|
7
|
+
/** OAuth access token sent as `Authorization: Bearer <token>`. */
|
|
8
|
+
token: string;
|
|
9
|
+
/** Cloud Code Assist endpoint. Defaults to `https://cloudcode-pa.googleapis.com`. */
|
|
10
|
+
endpoint?: string;
|
|
11
|
+
/** Pre-resolved GCP project id; otherwise discovered via `loadCodeAssist`. */
|
|
12
|
+
projectId?: string;
|
|
13
|
+
/** Optional abort signal for request cancellation. */
|
|
14
|
+
signal?: AbortSignal;
|
|
15
|
+
/** Optional fetch implementation override for tests. */
|
|
16
|
+
fetcher?: typeof fetch;
|
|
17
|
+
/** Effort-tier collapse table applied to the discovered list. */
|
|
18
|
+
collapseTable?: VariantCollapseTable;
|
|
19
|
+
}
|
|
20
|
+
/**
|
|
21
|
+
* Discovers the Gemini models available to a `google-gemini-cli` credential via
|
|
22
|
+
* the account's own `retrieveUserQuota` endpoint on Cloud Code Assist.
|
|
23
|
+
*
|
|
24
|
+
* This is the fallback for accounts whose credential is not authorized for the
|
|
25
|
+
* Antigravity `fetchAvailableModels` endpoint (e.g. Gemini Code Assist Standard
|
|
26
|
+
* tiers, which return HTTP 403 there). Quota buckets carry only model ids, so
|
|
27
|
+
* metadata is filled from the bundled catalog where the id is known and
|
|
28
|
+
* synthesized with Gemini CLI defaults otherwise.
|
|
29
|
+
*
|
|
30
|
+
* Returns `null` on network/payload/auth failure (the caller keeps the bundled
|
|
31
|
+
* catalog). Returns `[]` when the quota response lists no usable Gemini models.
|
|
32
|
+
*/
|
|
33
|
+
export declare function fetchGeminiCliQuotaModels(options: FetchGeminiCliQuotaModelsOptions): Promise<ModelSpec<"google-gemini-cli">[] | null>;
|
package/dist/types/hosts.d.ts
CHANGED
|
@@ -111,6 +111,11 @@ export declare const KNOWN_HOSTS: {
|
|
|
111
111
|
readonly providers: readonly ["nvidia"];
|
|
112
112
|
readonly urlMarkers: readonly ["integrate.api.nvidia.com"];
|
|
113
113
|
};
|
|
114
|
+
/** Venice AI (`api.venice.ai`). OpenAI-compatible; drives reasoning via top-level `reasoning_effort` (and `venice_parameters.disable_thinking`), and rejects DashScope's top-level `enable_thinking` with a 400 (`additionalProperties: false` request schema). */
|
|
115
|
+
readonly venice: {
|
|
116
|
+
readonly providers: readonly ["venice"];
|
|
117
|
+
readonly urlMarkers: readonly ["api.venice.ai"];
|
|
118
|
+
};
|
|
114
119
|
readonly moonshotNative: {
|
|
115
120
|
readonly providers: readonly ["moonshot", "kimi-code"];
|
|
116
121
|
readonly urlMarkers: readonly ["api.moonshot.ai", "api.kimi.com"];
|
|
@@ -212,7 +212,7 @@ export declare const supportsMidConversationSystemMessages: (modelId: string) =>
|
|
|
212
212
|
* Models that reliably follow the hashline line-anchored edit dialect
|
|
213
213
|
* (`[path#TAG]` headers plus 1-indexed anchors). Kimi, MiMo, DeepSeek V4
|
|
214
214
|
* Flash, and Step 3.7 Flash miscount anchors or drop the tag header often
|
|
215
|
-
* enough that hosts fall back to
|
|
215
|
+
* enough that hosts fall back to the sloppy edit format for
|
|
216
216
|
* them.
|
|
217
217
|
*/
|
|
218
218
|
export declare const supportsHashlineEdits: (modelId: string) => boolean;
|
|
@@ -18,6 +18,8 @@ export interface GoogleAntigravityModelManagerConfig {
|
|
|
18
18
|
}
|
|
19
19
|
export interface GoogleGeminiCliModelManagerConfig {
|
|
20
20
|
oauthToken?: string;
|
|
21
|
+
/** GCP project id required by Workspace/Standard credentials for quota discovery. */
|
|
22
|
+
projectId?: string;
|
|
21
23
|
endpoint?: string;
|
|
22
24
|
fetch?: FetchImpl;
|
|
23
25
|
}
|
package/dist/types/types.d.ts
CHANGED
|
@@ -122,7 +122,7 @@ export interface Usage {
|
|
|
122
122
|
};
|
|
123
123
|
}
|
|
124
124
|
export type OpenAIReasoningFormat = "openai" | "openrouter" | "zai" | "kimi" | "qwen" | "qwen-chat-template";
|
|
125
|
-
export type OpenAIReasoningDisableMode = "omit" | "lowest-effort" | "none-effort" | "openrouter-enabled-false" | "zai-thinking-disabled" | "qwen-enable-thinking-false" | "qwen-template-false";
|
|
125
|
+
export type OpenAIReasoningDisableMode = "omit" | "lowest-effort" | "none-effort" | "openrouter-enabled-false" | "venice-disable-thinking" | "zai-thinking-disabled" | "qwen-enable-thinking-false" | "qwen-template-false";
|
|
126
126
|
export type OpenAIStreamMarkupHealingPattern = "kimi" | "dsml" | "qwen" | "thinking";
|
|
127
127
|
/**
|
|
128
128
|
* Compatibility settings for openai-completions API.
|
|
@@ -727,6 +727,12 @@ export interface ModelCost extends TokenCost {
|
|
|
727
727
|
export type ModelTokenizer = "claude-v3" | "claude-v47" | "claude-v5" | "claude-v5-sonnet" | "qwen3" | "deepseek-v3" | "kimi-k2" | "glm5";
|
|
728
728
|
export interface Model<TApi extends Api = Api> {
|
|
729
729
|
id: string;
|
|
730
|
+
/**
|
|
731
|
+
* Whether provider-bound private-use glyphs require reversible ASCII tokenization.
|
|
732
|
+
* Materialized by `buildModel`; request handlers read this capability instead of
|
|
733
|
+
* inferring it from the transport API.
|
|
734
|
+
*/
|
|
735
|
+
requiresGlyphTokenization?: boolean;
|
|
730
736
|
/**
|
|
731
737
|
* Model id to send on the wire when it differs from `id`. Used by catalog
|
|
732
738
|
* variants that present one upstream model under several local entries —
|
|
@@ -846,13 +852,25 @@ export interface Model<TApi extends Api = Api> {
|
|
|
846
852
|
* `options.isOAuth = true` for the underlying provider call.
|
|
847
853
|
*/
|
|
848
854
|
isOAuth?: boolean;
|
|
855
|
+
/**
|
|
856
|
+
* Amazon Bedrock Guardrail id or ARN attached to every Converse request for
|
|
857
|
+
* this model. Set from `providers.amazon-bedrock.guardrailIdentifier`; the
|
|
858
|
+
* streaming layer forwards it as `options.guardrailIdentifier` so accounts
|
|
859
|
+
* that gate `bedrock:InvokeModel*` on the `bedrock:GuardrailIdentifier`
|
|
860
|
+
* condition key stop returning an explicit deny.
|
|
861
|
+
*/
|
|
862
|
+
guardrailIdentifier?: string;
|
|
863
|
+
/** Bedrock guardrail version. Defaults to `"DRAFT"` at request time when unset. */
|
|
864
|
+
guardrailVersion?: string;
|
|
865
|
+
/** Bedrock guardrail trace verbosity. */
|
|
866
|
+
guardrailTrace?: "enabled" | "disabled" | "enabled_full";
|
|
849
867
|
}
|
|
850
868
|
/**
|
|
851
869
|
* A model as authored by configs, bundled catalogs, and discovery — the input
|
|
852
870
|
* vocabulary of `buildModel`. Identical to `Model` except `compat` carries the
|
|
853
871
|
* sparse override shape and nothing is resolved yet.
|
|
854
872
|
*/
|
|
855
|
-
export interface ModelSpec<TApi extends Api = Api> extends Omit<Model<TApi>, "compat" | "compatConfig" | "supportsComputerUseConfig"> {
|
|
873
|
+
export interface ModelSpec<TApi extends Api = Api> extends Omit<Model<TApi>, "compat" | "compatConfig" | "requiresGlyphTokenization" | "supportsComputerUseConfig"> {
|
|
856
874
|
/** Sparse compatibility overrides; resolved into `Model.compat` by `buildModel`. */
|
|
857
875
|
compat?: CompatConfigOf<TApi>;
|
|
858
876
|
}
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@oh-my-pi/pi-catalog",
|
|
4
|
-
"version": "
|
|
4
|
+
"version": "18.0.1",
|
|
5
5
|
"description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence",
|
|
6
6
|
"homepage": "https://omp.sh",
|
|
7
7
|
"author": "Stencil Labs, Inc.",
|
|
@@ -34,11 +34,11 @@
|
|
|
34
34
|
"gen:proto": "bun scripts/generate-protocols.ts"
|
|
35
35
|
},
|
|
36
36
|
"dependencies": {
|
|
37
|
-
"@oh-my-pi/omptype": "
|
|
38
|
-
"@oh-my-pi/pi-utils": "
|
|
37
|
+
"@oh-my-pi/omptype": "18.0.1",
|
|
38
|
+
"@oh-my-pi/pi-utils": "18.0.1"
|
|
39
39
|
},
|
|
40
40
|
"devDependencies": {
|
|
41
|
-
"@oh-my-pi/pi-ai": "
|
|
41
|
+
"@oh-my-pi/pi-ai": "18.0.1",
|
|
42
42
|
"@types/bun": "^1.3.14"
|
|
43
43
|
},
|
|
44
44
|
"engines": {
|
package/src/build.ts
CHANGED
|
@@ -15,6 +15,7 @@ import { buildBedrockCompat } from "./compat/bedrock";
|
|
|
15
15
|
import { buildDevinCompat } from "./compat/devin";
|
|
16
16
|
import { buildOpenAICompat, buildOpenAIResponsesCompat, buildOpenRouterCompat } from "./compat/openai";
|
|
17
17
|
import { bareModelId, parseOpenAIModel, semverGte } from "./identity/classify";
|
|
18
|
+
import { isClaudeModelId } from "./identity/family";
|
|
18
19
|
import { resolveModelThinking } from "./model-thinking";
|
|
19
20
|
import { resolveModelTokenizer } from "./model-tokenizer";
|
|
20
21
|
import type { Api, CompatOf, Model, ModelSpec } from "./types";
|
|
@@ -70,6 +71,7 @@ export function buildModel<TApi extends Api>(spec: ModelSpec<TApi>): Model<TApi>
|
|
|
70
71
|
return {
|
|
71
72
|
...spec,
|
|
72
73
|
name: cleanModelName(spec.name),
|
|
74
|
+
requiresGlyphTokenization: isClaudeModelId(spec.id),
|
|
73
75
|
tokenizer: spec.tokenizer ?? resolveModelTokenizer(spec.requestModelId ?? spec.id),
|
|
74
76
|
thinking: resolveModelThinking(spec, compat),
|
|
75
77
|
supportsComputerUse: supportsOpenAIGAComputerUse(spec, supportsComputerUseConfig),
|
package/src/compat/openai.ts
CHANGED
|
@@ -312,6 +312,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
|
|
312
312
|
modelMatchesHost(hostModel, "anthropic") || isClaudeModelId(spec.id) || isAnthropicNamespacedModelId(spec.id);
|
|
313
313
|
const isAlibaba = modelMatchesHost(hostModel, "alibabaDashscope");
|
|
314
314
|
const isNvidiaNim = modelMatchesHost(hostModel, "nvidia");
|
|
315
|
+
const isVenice = modelMatchesHost(hostModel, "venice");
|
|
315
316
|
const isQwen = isQwenModelId(spec.id);
|
|
316
317
|
// DeepSeek V4 (and other reasoning-capable DeepSeek models) reject follow-up requests in
|
|
317
318
|
// thinking mode unless prior assistant tool-call turns include `reasoning_content`. The
|
|
@@ -467,7 +468,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
|
|
467
468
|
? "openrouter"
|
|
468
469
|
: isQwen && (isNvidiaNim || provider === "vllm")
|
|
469
470
|
? "qwen-chat-template"
|
|
470
|
-
: isQwen && isFireworks
|
|
471
|
+
: isQwen && (isFireworks || isVenice)
|
|
471
472
|
? "openai"
|
|
472
473
|
: isAlibaba || isQwen
|
|
473
474
|
? "qwen"
|
|
@@ -528,7 +529,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
|
|
528
529
|
// (issue #2299).
|
|
529
530
|
thinkingFormat,
|
|
530
531
|
kimiApiFormat: undefined,
|
|
531
|
-
reasoningDisableMode: resolveReasoningDisableMode(thinkingFormat),
|
|
532
|
+
reasoningDisableMode: isVenice ? "venice-disable-thinking" : resolveReasoningDisableMode(thinkingFormat),
|
|
532
533
|
omitReasoningEffort: false,
|
|
533
534
|
includeEncryptedReasoning: true,
|
|
534
535
|
filterReasoningHistory: isOpenRouter && isAnthropicModel,
|
|
@@ -648,7 +649,9 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
|
|
648
649
|
? "omit"
|
|
649
650
|
: isDirectDeepseekReasoning
|
|
650
651
|
? "zai-thinking-disabled"
|
|
651
|
-
:
|
|
652
|
+
: isVenice
|
|
653
|
+
? "venice-disable-thinking"
|
|
654
|
+
: resolveReasoningDisableMode(compat.thinkingFormat);
|
|
652
655
|
}
|
|
653
656
|
if (spec.compat?.omitReasoningEffort === undefined && !compat.supportsReasoningEffort) {
|
|
654
657
|
compat.omitReasoningEffort = true;
|
|
@@ -666,7 +669,9 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
|
|
666
669
|
const variant: ResolvedOpenAICompat = { ...compat };
|
|
667
670
|
applyCompatOverrides(variant, whenThinkingPolicy);
|
|
668
671
|
if (whenThinkingPolicy.reasoningDisableMode === undefined) {
|
|
669
|
-
variant.reasoningDisableMode =
|
|
672
|
+
variant.reasoningDisableMode = isVenice
|
|
673
|
+
? "venice-disable-thinking"
|
|
674
|
+
: resolveReasoningDisableMode(variant.thinkingFormat);
|
|
670
675
|
}
|
|
671
676
|
if (whenThinkingPolicy.omitReasoningEffort === undefined && !variant.supportsReasoningEffort) {
|
|
672
677
|
variant.omitReasoningEffort = true;
|
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
import { type } from "@oh-my-pi/omptype";
|
|
2
|
+
import type { FetchImpl } from "@oh-my-pi/pi-utils";
|
|
3
|
+
import { parseGeminiModel, semverGte } from "../identity/classify";
|
|
4
|
+
import { isGeminiModelId } from "../identity/family";
|
|
5
|
+
import { createBundledReferenceMap } from "../provider-models/bundled-references";
|
|
6
|
+
import type { ModelSpec } from "../types";
|
|
7
|
+
import { discoveryFetch } from "../utils";
|
|
8
|
+
import {
|
|
9
|
+
collapseEffortVariants,
|
|
10
|
+
GEMINI_CLI_VARIANT_COLLAPSE_TABLE,
|
|
11
|
+
type VariantCollapseTable,
|
|
12
|
+
} from "../variant-collapse";
|
|
13
|
+
import { getGeminiCliHeaders } from "../wire/gemini-headers";
|
|
14
|
+
|
|
15
|
+
const DEFAULT_ENDPOINT = "https://cloudcode-pa.googleapis.com";
|
|
16
|
+
const LOAD_CODE_ASSIST_PATH = "/v1internal:loadCodeAssist";
|
|
17
|
+
const RETRIEVE_USER_QUOTA_PATH = "/v1internal:retrieveUserQuota";
|
|
18
|
+
|
|
19
|
+
// All current Gemini CLI models ship a 1M-token context and 65,536-token
|
|
20
|
+
// output ceiling; used only for quota-listed ids the bundled catalog does not
|
|
21
|
+
// already describe. Ids present in the bundle keep their real limits.
|
|
22
|
+
const DEFAULT_CONTEXT_WINDOW = 1_048_576;
|
|
23
|
+
const DEFAULT_MAX_TOKENS = 65_536;
|
|
24
|
+
|
|
25
|
+
/** Gemini generations that expose thinking on Cloud Code Assist. */
|
|
26
|
+
const REASONING_MIN_VERSION = "2.5";
|
|
27
|
+
|
|
28
|
+
const LoadCodeAssistResponseSchema = type({
|
|
29
|
+
"cloudaicompanionProject?": type("unknown").pipe(value => {
|
|
30
|
+
if (typeof value === "string") return value;
|
|
31
|
+
if (value && typeof value === "object" && "id" in value && typeof value.id === "string") {
|
|
32
|
+
return value.id;
|
|
33
|
+
}
|
|
34
|
+
return undefined;
|
|
35
|
+
}),
|
|
36
|
+
});
|
|
37
|
+
|
|
38
|
+
const QuotaBucketSchema = type({
|
|
39
|
+
"modelId?": type("unknown").pipe(value => (typeof value === "string" ? value : undefined)),
|
|
40
|
+
});
|
|
41
|
+
|
|
42
|
+
const RetrieveUserQuotaResponseSchema = type({
|
|
43
|
+
"buckets?": type("unknown").pipe(value => {
|
|
44
|
+
if (!Array.isArray(value)) return undefined;
|
|
45
|
+
const buckets: Array<{ modelId?: string }> = [];
|
|
46
|
+
for (const bucket of value) {
|
|
47
|
+
const parsed = QuotaBucketSchema(bucket);
|
|
48
|
+
if (!(parsed instanceof type.errors)) {
|
|
49
|
+
buckets.push(parsed);
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
return buckets;
|
|
53
|
+
}),
|
|
54
|
+
});
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Options for the Gemini CLI quota-based discovery fallback.
|
|
58
|
+
*/
|
|
59
|
+
export interface FetchGeminiCliQuotaModelsOptions {
|
|
60
|
+
/** OAuth access token sent as `Authorization: Bearer <token>`. */
|
|
61
|
+
token: string;
|
|
62
|
+
/** Cloud Code Assist endpoint. Defaults to `https://cloudcode-pa.googleapis.com`. */
|
|
63
|
+
endpoint?: string;
|
|
64
|
+
/** Pre-resolved GCP project id; otherwise discovered via `loadCodeAssist`. */
|
|
65
|
+
projectId?: string;
|
|
66
|
+
/** Optional abort signal for request cancellation. */
|
|
67
|
+
signal?: AbortSignal;
|
|
68
|
+
/** Optional fetch implementation override for tests. */
|
|
69
|
+
fetcher?: typeof fetch;
|
|
70
|
+
/** Effort-tier collapse table applied to the discovered list. */
|
|
71
|
+
collapseTable?: VariantCollapseTable;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* Discovers the Gemini models available to a `google-gemini-cli` credential via
|
|
76
|
+
* the account's own `retrieveUserQuota` endpoint on Cloud Code Assist.
|
|
77
|
+
*
|
|
78
|
+
* This is the fallback for accounts whose credential is not authorized for the
|
|
79
|
+
* Antigravity `fetchAvailableModels` endpoint (e.g. Gemini Code Assist Standard
|
|
80
|
+
* tiers, which return HTTP 403 there). Quota buckets carry only model ids, so
|
|
81
|
+
* metadata is filled from the bundled catalog where the id is known and
|
|
82
|
+
* synthesized with Gemini CLI defaults otherwise.
|
|
83
|
+
*
|
|
84
|
+
* Returns `null` on network/payload/auth failure (the caller keeps the bundled
|
|
85
|
+
* catalog). Returns `[]` when the quota response lists no usable Gemini models.
|
|
86
|
+
*/
|
|
87
|
+
export async function fetchGeminiCliQuotaModels(
|
|
88
|
+
options: FetchGeminiCliQuotaModelsOptions,
|
|
89
|
+
): Promise<ModelSpec<"google-gemini-cli">[] | null> {
|
|
90
|
+
const fetcher = discoveryFetch(options.fetcher);
|
|
91
|
+
const endpoint = (options.endpoint?.trim() || DEFAULT_ENDPOINT).replace(/\/+$/, "");
|
|
92
|
+
const headers = {
|
|
93
|
+
Authorization: `Bearer ${options.token}`,
|
|
94
|
+
"Content-Type": "application/json",
|
|
95
|
+
...getGeminiCliHeaders(),
|
|
96
|
+
};
|
|
97
|
+
|
|
98
|
+
const projectId = options.projectId ?? (await loadProjectId(fetcher, endpoint, headers, options.signal));
|
|
99
|
+
|
|
100
|
+
let response: Response;
|
|
101
|
+
try {
|
|
102
|
+
response = await fetcher(`${endpoint}${RETRIEVE_USER_QUOTA_PATH}`, {
|
|
103
|
+
method: "POST",
|
|
104
|
+
headers,
|
|
105
|
+
body: JSON.stringify(projectId ? { project: projectId } : {}),
|
|
106
|
+
signal: options.signal,
|
|
107
|
+
});
|
|
108
|
+
} catch {
|
|
109
|
+
return null;
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
if (!response.ok) {
|
|
113
|
+
return null;
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
let payload: unknown;
|
|
117
|
+
try {
|
|
118
|
+
payload = await response.json();
|
|
119
|
+
} catch {
|
|
120
|
+
return null;
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
const parsed = RetrieveUserQuotaResponseSchema(payload);
|
|
124
|
+
if (parsed instanceof type.errors) {
|
|
125
|
+
return null;
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
const seen = new Set<string>();
|
|
129
|
+
const models: ModelSpec<"google-gemini-cli">[] = [];
|
|
130
|
+
const bundled = createBundledReferenceMap<"google-gemini-cli">("google-gemini-cli");
|
|
131
|
+
|
|
132
|
+
for (const bucket of parsed.buckets ?? []) {
|
|
133
|
+
const modelId = bucket.modelId?.trim();
|
|
134
|
+
if (!modelId || seen.has(modelId) || !isGeminiModelId(modelId)) {
|
|
135
|
+
continue;
|
|
136
|
+
}
|
|
137
|
+
seen.add(modelId);
|
|
138
|
+
|
|
139
|
+
const reference = bundled.get(modelId);
|
|
140
|
+
if (reference) {
|
|
141
|
+
models.push({ ...reference, baseUrl: endpoint });
|
|
142
|
+
continue;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
const parsedId = parseGeminiModel(modelId);
|
|
146
|
+
models.push({
|
|
147
|
+
id: modelId,
|
|
148
|
+
name: modelId,
|
|
149
|
+
api: "google-gemini-cli",
|
|
150
|
+
provider: "google-gemini-cli",
|
|
151
|
+
baseUrl: endpoint,
|
|
152
|
+
reasoning: parsedId ? semverGte(parsedId.version, REASONING_MIN_VERSION) : false,
|
|
153
|
+
input: ["text", "image"],
|
|
154
|
+
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
|
155
|
+
contextWindow: DEFAULT_CONTEXT_WINDOW,
|
|
156
|
+
maxTokens: DEFAULT_MAX_TOKENS,
|
|
157
|
+
});
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
const collapsed = collapseEffortVariants(models, options.collapseTable ?? GEMINI_CLI_VARIANT_COLLAPSE_TABLE);
|
|
161
|
+
collapsed.sort((a, b) => a.name.localeCompare(b.name) || a.id.localeCompare(b.id));
|
|
162
|
+
return collapsed;
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
async function loadProjectId(
|
|
166
|
+
fetcher: FetchImpl,
|
|
167
|
+
endpoint: string,
|
|
168
|
+
headers: Record<string, string>,
|
|
169
|
+
signal: AbortSignal | undefined,
|
|
170
|
+
): Promise<string | undefined> {
|
|
171
|
+
let response: Response;
|
|
172
|
+
try {
|
|
173
|
+
response = await fetcher(`${endpoint}${LOAD_CODE_ASSIST_PATH}`, {
|
|
174
|
+
method: "POST",
|
|
175
|
+
headers,
|
|
176
|
+
body: JSON.stringify({
|
|
177
|
+
metadata: { ideType: "IDE_UNSPECIFIED", platform: "PLATFORM_UNSPECIFIED", pluginType: "GEMINI" },
|
|
178
|
+
}),
|
|
179
|
+
signal,
|
|
180
|
+
});
|
|
181
|
+
} catch {
|
|
182
|
+
return undefined;
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
if (!response.ok) {
|
|
186
|
+
return undefined;
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
let payload: unknown;
|
|
190
|
+
try {
|
|
191
|
+
payload = await response.json();
|
|
192
|
+
} catch {
|
|
193
|
+
return undefined;
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
const parsed = LoadCodeAssistResponseSchema(payload);
|
|
197
|
+
return parsed instanceof type.errors ? undefined : parsed.cloudaicompanionProject;
|
|
198
|
+
}
|
package/src/discovery/index.ts
CHANGED
|
@@ -374,7 +374,7 @@ function compileMapField(desc: MapFieldDesc): CompiledField {
|
|
|
374
374
|
return {
|
|
375
375
|
number,
|
|
376
376
|
initDefault(message) {
|
|
377
|
-
Reflect.set(message, name,
|
|
377
|
+
Reflect.set(message, name, Object.create(null));
|
|
378
378
|
},
|
|
379
379
|
encode(message, writer) {
|
|
380
380
|
const input = Reflect.get(message, name);
|
|
@@ -425,7 +425,7 @@ function compileMapField(desc: MapFieldDesc): CompiledField {
|
|
|
425
425
|
toJson(message, output) {
|
|
426
426
|
const input = Reflect.get(message, name);
|
|
427
427
|
if (!isMessageObject(input)) return;
|
|
428
|
-
const mapOutput: { [key: string]: JsonValue } =
|
|
428
|
+
const mapOutput: { [key: string]: JsonValue } = Object.create(null);
|
|
429
429
|
for (const entryKey in input) {
|
|
430
430
|
mapOutput[entryKey] = valCodec.toJson(input[entryKey]);
|
|
431
431
|
}
|
|
@@ -685,7 +685,7 @@ function arrayField(message: object, name: string): unknown[] {
|
|
|
685
685
|
function mapField(message: object, name: string): Record<string, unknown> {
|
|
686
686
|
const value = Reflect.get(message, name);
|
|
687
687
|
if (isRecord(value)) return value;
|
|
688
|
-
const map: Record<string, unknown> =
|
|
688
|
+
const map: Record<string, unknown> = Object.create(null);
|
|
689
689
|
Reflect.set(message, name, map);
|
|
690
690
|
return map;
|
|
691
691
|
}
|
package/src/hosts.ts
CHANGED
|
@@ -61,6 +61,8 @@ export const KNOWN_HOSTS = {
|
|
|
61
61
|
qwenPortal: { providers: ["qwen-portal"], urlMarkers: ["portal.qwen.ai"] },
|
|
62
62
|
/** NVIDIA NIM (`integrate.api.nvidia.com`). Qwen NIM endpoints take `chat_template_kwargs.enable_thinking`, never top-level `enable_thinking`. */
|
|
63
63
|
nvidia: { providers: ["nvidia"], urlMarkers: ["integrate.api.nvidia.com"] },
|
|
64
|
+
/** Venice AI (`api.venice.ai`). OpenAI-compatible; drives reasoning via top-level `reasoning_effort` (and `venice_parameters.disable_thinking`), and rejects DashScope's top-level `enable_thinking` with a 400 (`additionalProperties: false` request schema). */
|
|
65
|
+
venice: { providers: ["venice"], urlMarkers: ["api.venice.ai"] },
|
|
64
66
|
moonshotNative: { providers: ["moonshot", "kimi-code"], urlMarkers: ["api.moonshot.ai", "api.kimi.com"] },
|
|
65
67
|
/** Google AI Studio's OpenAI-compatible shim (`/v1beta/openai`) — a subset of chat-completions; rejects `store` with a 400. Native Gemini uses `google-generative-ai` api instead. */
|
|
66
68
|
googleAistudio: { providers: [], urlMarkers: ["generativelanguage.googleapis.com"] },
|
package/src/identity/family.ts
CHANGED
|
@@ -418,7 +418,7 @@ export const supportsMidConversationSystemMessages = memo((modelId: string): boo
|
|
|
418
418
|
* Models that reliably follow the hashline line-anchored edit dialect
|
|
419
419
|
* (`[path#TAG]` headers plus 1-indexed anchors). Kimi, MiMo, DeepSeek V4
|
|
420
420
|
* Flash, and Step 3.7 Flash miscount anchors or drop the tag header often
|
|
421
|
-
* enough that hosts fall back to
|
|
421
|
+
* enough that hosts fall back to the sloppy edit format for
|
|
422
422
|
* them.
|
|
423
423
|
*/
|
|
424
424
|
export const supportsHashlineEdits = memo((modelId: string): boolean => {
|
package/src/model-thinking.ts
CHANGED
|
@@ -33,6 +33,7 @@ import {
|
|
|
33
33
|
isMinimaxM2FamilyModelId,
|
|
34
34
|
isMinimaxM3FamilyModelId,
|
|
35
35
|
isOpenAIGptOssModelId,
|
|
36
|
+
isQwenModelId,
|
|
36
37
|
supportsAdaptiveThinkingDisplay,
|
|
37
38
|
} from "./identity/family";
|
|
38
39
|
import type {
|
|
@@ -186,7 +187,9 @@ function fillThinkingWireDefaults<TApi extends Api>(
|
|
|
186
187
|
supportsAdaptiveThinkingDisplay(spec.id);
|
|
187
188
|
const needsRequiresEffort =
|
|
188
189
|
thinking.requiresEffort === undefined &&
|
|
189
|
-
(impliesMandatoryReasoning(parsed, spec.id) ||
|
|
190
|
+
(impliesMandatoryReasoning(parsed, spec.id) ||
|
|
191
|
+
isQwenTemplateReasoningEffortCompat(compat) ||
|
|
192
|
+
isOpenCodeGatewayOxAlphaModel(spec));
|
|
190
193
|
const needsDefaultLevel =
|
|
191
194
|
thinking.defaultLevel === undefined && (isKimiK3ModelId(spec.id) || isGlm53ReasoningEffortModelId(spec.id));
|
|
192
195
|
if (!effortsChanged && !shouldReplaceEffortMap && !needsDisplay && !needsRequiresEffort && !needsDefaultLevel) {
|
|
@@ -239,7 +242,11 @@ export function deriveThinking<TApi extends Api>(spec: ModelSpec<TApi>, compat:
|
|
|
239
242
|
) {
|
|
240
243
|
config.supportsDisplay = true;
|
|
241
244
|
}
|
|
242
|
-
if (
|
|
245
|
+
if (
|
|
246
|
+
impliesMandatoryReasoning(parsed, spec.id) ||
|
|
247
|
+
isQwenTemplateReasoningEffortCompat(compat) ||
|
|
248
|
+
isOpenCodeGatewayOxAlphaModel(spec)
|
|
249
|
+
) {
|
|
243
250
|
config.requiresEffort = true;
|
|
244
251
|
}
|
|
245
252
|
return config;
|
|
@@ -359,6 +366,9 @@ function getModelDefinedEfforts<TApi extends Api>(
|
|
|
359
366
|
if (isKimiK3ModelId(spec.id)) {
|
|
360
367
|
return LOW_HIGH_MAX_REASONING_EFFORTS;
|
|
361
368
|
}
|
|
369
|
+
if (isOpenCodeGatewayOxAlphaModel(spec)) {
|
|
370
|
+
return LOW_HIGH_MAX_REASONING_EFFORTS;
|
|
371
|
+
}
|
|
362
372
|
if (isSakanaFuguReasoningModel(spec)) {
|
|
363
373
|
return HIGH_MAX_REASONING_EFFORTS;
|
|
364
374
|
}
|
|
@@ -391,9 +401,15 @@ function getModelDefinedEfforts<TApi extends Api>(
|
|
|
391
401
|
return QWEN38_TEMPLATE_REASONING_EFFORTS;
|
|
392
402
|
}
|
|
393
403
|
if (
|
|
394
|
-
(isOpenAICompatReasoningApi(spec.api) ||
|
|
404
|
+
(isOpenAICompatReasoningApi(spec.api) ||
|
|
405
|
+
spec.api === "openai-responses" ||
|
|
406
|
+
(spec.api === "ollama-chat" && spec.provider === "ollama-cloud")) &&
|
|
395
407
|
isDeepseekReasoningModel(spec)
|
|
396
408
|
) {
|
|
409
|
+
// The DeepSeek V4 effort ladder is a model property, not a transport one:
|
|
410
|
+
// `opencode-go/deepseek-v4-flash` is pinned to `openai-responses` (the Go
|
|
411
|
+
// gateway serves it only at /responses), yet carries the same wire-exact
|
|
412
|
+
// low/high/max scale — so the Responses transport is admitted here too.
|
|
397
413
|
// DeepSeek V4 (Flash and Pro) accepts the wire-exact low/high/max ladder
|
|
398
414
|
// on every first-party/aggregator host — the direct API, aggregators, and
|
|
399
415
|
// Ollama Cloud alike (medium/xhigh fold into high, max is a real wire
|
|
@@ -538,6 +554,22 @@ function isSakanaFuguReasoningModel<TApi extends Api>(spec: ModelSpec<TApi>): bo
|
|
|
538
554
|
return spec.provider === "sakana" && /^fugu(?:$|-)/i.test(spec.id);
|
|
539
555
|
}
|
|
540
556
|
|
|
557
|
+
/**
|
|
558
|
+
* "Ox Alpha" stealth models on the OpenCode gateways (`opencode-go` /
|
|
559
|
+
* `opencode-zen`) reason through the wire-exact `low`/`high`/`max` ladder with
|
|
560
|
+
* mandatory thinking: the gateway rejects `minimal`/`medium`/`xhigh`
|
|
561
|
+
* (`[1210] ... please use low, high, or max`), the same dialect it already
|
|
562
|
+
* serves for GLM-5.3 and Kimi K3. Other hosts proxying an `ox-alpha` SKU
|
|
563
|
+
* (Kilo, NanoGPT, Venice, OpenRouter) expose their own vocabularies and are
|
|
564
|
+
* left untouched. See issue #9349.
|
|
565
|
+
*/
|
|
566
|
+
function isOpenCodeGatewayOxAlphaModel<TApi extends Api>(spec: ModelSpec<TApi>): boolean {
|
|
567
|
+
return (
|
|
568
|
+
(spec.provider === "opencode-go" || spec.provider === "opencode-zen") &&
|
|
569
|
+
/(?:^|\/)ox-alpha(?:-|$)/i.test(bareModelId(spec.id))
|
|
570
|
+
);
|
|
571
|
+
}
|
|
572
|
+
|
|
541
573
|
function isDeepseekReasoningModel<TApi extends Api>(spec: ModelSpec<TApi>): boolean {
|
|
542
574
|
if (!spec.reasoning) return false;
|
|
543
575
|
const lowerId = spec.id.toLowerCase();
|
|
@@ -660,6 +692,13 @@ function inferFallbackEfforts<TApi extends Api>(spec: ModelSpec<TApi>, compat: C
|
|
|
660
692
|
}
|
|
661
693
|
if (isOpenAICompatReasoningApi(spec.api)) {
|
|
662
694
|
const resolved = compat as ResolvedOpenAICompat;
|
|
695
|
+
if (
|
|
696
|
+
resolved.thinkingFormat === "openai" &&
|
|
697
|
+
modelMatchesHost({ provider: spec.provider, baseUrl: spec.baseUrl ?? "" }, "venice") &&
|
|
698
|
+
isQwenModelId(spec.id)
|
|
699
|
+
) {
|
|
700
|
+
return DEFAULT_REASONING_EFFORTS;
|
|
701
|
+
}
|
|
663
702
|
if (resolved.thinkingFormat === "openai" && resolved.supportsReasoningEffort) {
|
|
664
703
|
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
|
665
704
|
}
|