@oh-my-pi/pi-catalog 17.2.1 → 17.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +15 -0
- package/README.md +13 -13
- package/dist/types/hosts.d.ts +5 -0
- package/dist/types/model-manager.d.ts +5 -5
- package/dist/types/provider-models/descriptors.d.ts +9 -0
- package/dist/types/provider-models/openai-compat.d.ts +31 -0
- package/dist/types/types.d.ts +2 -0
- package/package.json +3 -3
- package/src/compat/openai.ts +4 -0
- package/src/discovery/cursor.ts +59 -2
- package/src/hosts.ts +2 -0
- package/src/model-manager.ts +34 -8
- package/src/models.json +6073 -609
- package/src/models.ts +1 -1
- package/src/provider-models/cache-provider-id.ts +3 -1
- package/src/provider-models/descriptors.ts +9 -0
- package/src/provider-models/openai-compat.ts +266 -19
- package/src/types.ts +2 -0
package/src/models.ts
CHANGED
|
@@ -6,7 +6,7 @@ import type { Api, KnownProvider, Model, ModelSpec, Usage } from "./types";
|
|
|
6
6
|
* Static bundled model registry loaded from `models.json`.
|
|
7
7
|
*
|
|
8
8
|
* This module intentionally exposes compile-time defaults only.
|
|
9
|
-
* It does not include runtime discovery,
|
|
9
|
+
* It does not include runtime discovery, stencil.so overlays, or on-disk cache state.
|
|
10
10
|
*
|
|
11
11
|
* For runtime-aware resolution, use `createModelManager()` / `resolveProviderModels()`.
|
|
12
12
|
*/
|
|
@@ -41,7 +41,9 @@ export function resolveModelCacheProviderId(providerId: string, options: ModelCa
|
|
|
41
41
|
case "ollama":
|
|
42
42
|
return resolveOllamaModelCacheProviderId(providerId, options.baseUrl);
|
|
43
43
|
case "cursor":
|
|
44
|
-
|
|
44
|
+
// v3: max-mode Claude/Gemini rows cached before the 1M context-window
|
|
45
|
+
// discovery fix carry a stale 200k window and must be refetched.
|
|
46
|
+
return "cursor:max-mode-v3";
|
|
45
47
|
case "litellm": {
|
|
46
48
|
const baseUrl = options.baseUrl ?? getDefaultModelDiscoveryBaseUrl(providerId)!;
|
|
47
49
|
return `litellm:rich-v5:${Bun.hash(baseUrl).toString(36)}`;
|
|
@@ -21,6 +21,7 @@ import {
|
|
|
21
21
|
firepassModelManagerOptions,
|
|
22
22
|
fireworksModelManagerOptions,
|
|
23
23
|
githubCopilotModelManagerOptions,
|
|
24
|
+
gmiCloudModelManagerOptions,
|
|
24
25
|
groqModelManagerOptions,
|
|
25
26
|
huggingfaceModelManagerOptions,
|
|
26
27
|
kiloModelManagerOptions,
|
|
@@ -178,6 +179,14 @@ export const CATALOG_PROVIDERS = [
|
|
|
178
179
|
createModelManagerOptions: (config: ModelManagerConfig) => gitLabDuoWorkflowModelManagerOptions(config),
|
|
179
180
|
dynamicModelsAuthoritative: true,
|
|
180
181
|
},
|
|
182
|
+
{
|
|
183
|
+
id: "gmi-cloud",
|
|
184
|
+
defaultModel: "deepseek-ai/DeepSeek-V4-Flash",
|
|
185
|
+
envVars: ["GMI_API_KEY"],
|
|
186
|
+
createModelManagerOptions: (config: ModelManagerConfig) => gmiCloudModelManagerOptions(config),
|
|
187
|
+
dynamicModelsAuthoritative: true,
|
|
188
|
+
catalogDiscovery: { label: "GMI Cloud" },
|
|
189
|
+
},
|
|
181
190
|
{
|
|
182
191
|
id: "google",
|
|
183
192
|
defaultModel: "gemini-3.1-pro-preview",
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { VERSION } from "@oh-my-pi/pi-utils";
|
|
1
2
|
import * as logger from "@oh-my-pi/pi-utils/logger";
|
|
2
3
|
import {
|
|
3
4
|
fetchOpenAICompatibleModels,
|
|
@@ -31,7 +32,10 @@ import {
|
|
|
31
32
|
import { createBundledReferenceMap, createReferenceResolver, toModelSpec } from "./bundled-references";
|
|
32
33
|
import { getDefaultModelDiscoveryBaseUrl, resolveModelCacheProviderId } from "./cache-provider-id";
|
|
33
34
|
|
|
34
|
-
const MODELS_DEV_URL = "https://
|
|
35
|
+
const MODELS_DEV_URL = "https://catalog.stencil.so/models.json.zstd";
|
|
36
|
+
|
|
37
|
+
/** Little-endian magic number opening every zstd frame (RFC 8878). */
|
|
38
|
+
const ZSTD_MAGIC = 0xfd2fb528;
|
|
35
39
|
|
|
36
40
|
/**
|
|
37
41
|
* Uses a cancellable timer rather than the native abort-timeout helper so
|
|
@@ -93,16 +97,74 @@ function toInputCapabilities(value: unknown): ("text" | "image")[] {
|
|
|
93
97
|
return supportsImage ? ["text", "image"] : ["text"];
|
|
94
98
|
}
|
|
95
99
|
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
100
|
+
/**
|
|
101
|
+
* Process-wide catalog session: the first call downloads the payload (the one
|
|
102
|
+
* request the server logs); later calls revalidate with `If-None-Match` and
|
|
103
|
+
* reuse the decoded payload on `304`. Failure after a successful load falls
|
|
104
|
+
* back to the session copy.
|
|
105
|
+
*/
|
|
106
|
+
const catalogSession: {
|
|
107
|
+
inflight: Promise<unknown> | null;
|
|
108
|
+
payload: unknown;
|
|
109
|
+
etag: string | null;
|
|
110
|
+
hasPayload: boolean;
|
|
111
|
+
} = { inflight: null, payload: undefined, etag: null, hasPayload: false };
|
|
112
|
+
|
|
113
|
+
const CATALOG_USER_AGENT = `omp/${VERSION} (+https://omp.sh)`;
|
|
114
|
+
|
|
115
|
+
/**
|
|
116
|
+
* Fetches the models.dev catalog via catalog.stencil.so, which serves a
|
|
117
|
+
* field-pruned copy precompressed as a zstd blob (~93 KB vs ~3.3 MB raw).
|
|
118
|
+
* The frame magic is sniffed rather than trusting content-type so plain-JSON
|
|
119
|
+
* responses (test stubs, fallback mirrors) parse identically.
|
|
120
|
+
*
|
|
121
|
+
* Fetched fully once per process: concurrent callers share the in-flight
|
|
122
|
+
* request, repeat callers send a conditional GET that the server answers
|
|
123
|
+
* (and deliberately does not log) with `304`.
|
|
124
|
+
*/
|
|
125
|
+
export function fetchWellKnownModels(fetchImpl?: FetchImpl, signal?: AbortSignal): Promise<unknown> {
|
|
126
|
+
if (!catalogSession.inflight) {
|
|
127
|
+
catalogSession.inflight = fetchCatalogPayload(fetchImpl ?? discoveryFetch(), signal).finally(() => {
|
|
128
|
+
catalogSession.inflight = null;
|
|
129
|
+
});
|
|
130
|
+
}
|
|
131
|
+
return catalogSession.inflight;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
async function fetchCatalogPayload(fetchImpl: FetchImpl, signal?: AbortSignal): Promise<unknown> {
|
|
135
|
+
const headers: Record<string, string> = {
|
|
136
|
+
Accept: "application/zstd, application/json",
|
|
137
|
+
"User-Agent": CATALOG_USER_AGENT,
|
|
138
|
+
};
|
|
139
|
+
if (catalogSession.hasPayload && catalogSession.etag) {
|
|
140
|
+
headers["If-None-Match"] = catalogSession.etag;
|
|
141
|
+
}
|
|
142
|
+
let response: Response;
|
|
143
|
+
try {
|
|
144
|
+
response = await fetchImpl(MODELS_DEV_URL, { method: "GET", headers, signal });
|
|
145
|
+
} catch (error) {
|
|
146
|
+
if (catalogSession.hasPayload) {
|
|
147
|
+
return catalogSession.payload;
|
|
148
|
+
}
|
|
149
|
+
throw error;
|
|
150
|
+
}
|
|
151
|
+
if (response.status === 304 && catalogSession.hasPayload) {
|
|
152
|
+
return catalogSession.payload;
|
|
153
|
+
}
|
|
102
154
|
if (!response.ok) {
|
|
103
|
-
|
|
155
|
+
if (catalogSession.hasPayload) {
|
|
156
|
+
return catalogSession.payload;
|
|
157
|
+
}
|
|
158
|
+
throw new Error(`models catalog fetch failed: ${response.status}`);
|
|
104
159
|
}
|
|
105
|
-
|
|
160
|
+
const bytes = new Uint8Array(await response.arrayBuffer());
|
|
161
|
+
const isZstd = bytes.length >= 4 && new DataView(bytes.buffer, bytes.byteOffset).getUint32(0, true) === ZSTD_MAGIC;
|
|
162
|
+
const text = new TextDecoder().decode(isZstd ? await Bun.zstdDecompress(bytes) : bytes);
|
|
163
|
+
const payload: unknown = JSON.parse(text);
|
|
164
|
+
catalogSession.payload = payload;
|
|
165
|
+
catalogSession.etag = response.headers.get("etag");
|
|
166
|
+
catalogSession.hasPayload = true;
|
|
167
|
+
return payload;
|
|
106
168
|
}
|
|
107
169
|
|
|
108
170
|
function mapAnthropicModelsDev(payload: unknown, baseUrl: string): ModelSpec<"anthropic-messages">[] {
|
|
@@ -887,6 +949,53 @@ export function projectOpenAIProReasoningAliases(models: readonly ModelSpec<Api>
|
|
|
887
949
|
return out;
|
|
888
950
|
}
|
|
889
951
|
|
|
952
|
+
// ---------------------------------------------------------------------------
|
|
953
|
+
// 1b. GMI Cloud
|
|
954
|
+
// ---------------------------------------------------------------------------
|
|
955
|
+
|
|
956
|
+
const GMI_CLOUD_BASE_URL = "https://api.gmi-serving.com/v1";
|
|
957
|
+
|
|
958
|
+
/**
|
|
959
|
+
* Bundled seed for GMI Cloud. Generation has no `GMI_API_KEY`, so a regen
|
|
960
|
+
* without credentials would leave the provider slice empty and the declared
|
|
961
|
+
* `defaultModel` unresolvable on a fresh install before the async runtime
|
|
962
|
+
* discovery fires. Live `/v1/models` discovery is authoritative for the model
|
|
963
|
+
* ID set and overrides context/max-token limits, but `mapWithBundledReference`
|
|
964
|
+
* keeps the reference's cost/reasoning/thinking — so these fields carry GMI's
|
|
965
|
+
* direct-tariff values: V4-Flash at $0.14/$0.28 per 1M with Think High/Max
|
|
966
|
+
* modes per GMI's launch post
|
|
967
|
+
* (https://www.gmicloud.ai/en/blog/deepseek-v4-is-here-we-tested-it), not
|
|
968
|
+
* discounted gateway-route pricing. GMI publishes no cache-read tariff, so
|
|
969
|
+
* cacheRead stays 0 until a direct source confirms cached-token billing.
|
|
970
|
+
*/
|
|
971
|
+
export const GMI_CLOUD_STATIC_MODELS: readonly ModelSpec<"openai-completions">[] = [
|
|
972
|
+
{
|
|
973
|
+
id: "deepseek-ai/DeepSeek-V4-Flash",
|
|
974
|
+
name: "DeepSeek V4 Flash",
|
|
975
|
+
api: "openai-completions",
|
|
976
|
+
provider: "gmi-cloud",
|
|
977
|
+
baseUrl: GMI_CLOUD_BASE_URL,
|
|
978
|
+
reasoning: true,
|
|
979
|
+
input: ["text"],
|
|
980
|
+
cost: { input: 0.14, output: 0.28, cacheRead: 0, cacheWrite: 0 },
|
|
981
|
+
contextWindow: 1048576,
|
|
982
|
+
maxTokens: 384000,
|
|
983
|
+
thinking: { mode: "effort", efforts: [Effort.High, Effort.Max] },
|
|
984
|
+
},
|
|
985
|
+
];
|
|
986
|
+
|
|
987
|
+
export interface GmiCloudModelManagerConfig {
|
|
988
|
+
apiKey?: string;
|
|
989
|
+
baseUrl?: string;
|
|
990
|
+
fetch?: FetchImpl;
|
|
991
|
+
}
|
|
992
|
+
|
|
993
|
+
export function gmiCloudModelManagerOptions(
|
|
994
|
+
config?: GmiCloudModelManagerConfig,
|
|
995
|
+
): ModelManagerOptions<"openai-completions"> {
|
|
996
|
+
return createSimpleOpenAICompletionsOptions("gmi-cloud", GMI_CLOUD_BASE_URL, config);
|
|
997
|
+
}
|
|
998
|
+
|
|
890
999
|
// ---------------------------------------------------------------------------
|
|
891
1000
|
// 2. Groq
|
|
892
1001
|
// ---------------------------------------------------------------------------
|
|
@@ -1519,7 +1628,7 @@ async function loadSiliconFlowModelsDevReferences(
|
|
|
1519
1628
|
// Bounded: this enrichment is optional, so a stalled models.dev must not
|
|
1520
1629
|
// hold back the authoritative endpoint request that runs after it.
|
|
1521
1630
|
const payload = await withCatalogDiscoveryTimeout(SILICONFLOW_MODELS_DEV_REFERENCE_TIMEOUT_MS, signal =>
|
|
1522
|
-
|
|
1631
|
+
fetchWellKnownModels(fetchImpl, signal),
|
|
1523
1632
|
);
|
|
1524
1633
|
return createModelsDevReferenceMap<"openai-completions">(
|
|
1525
1634
|
mapModelsDevToModels(payload as Record<string, unknown>, [descriptor]),
|
|
@@ -1977,7 +2086,7 @@ function createModelsDevReferenceMap<TApi extends Api>(
|
|
|
1977
2086
|
|
|
1978
2087
|
async function loadModelsDevReferences<TApi extends Api>(fetchImpl?: FetchImpl): Promise<Map<string, ModelSpec<TApi>>> {
|
|
1979
2088
|
try {
|
|
1980
|
-
const payload = await
|
|
2089
|
+
const payload = await fetchWellKnownModels(fetchImpl);
|
|
1981
2090
|
return createModelsDevReferenceMap<TApi>(
|
|
1982
2091
|
mapModelsDevToModels(payload as Record<string, unknown>, MODELS_DEV_PROVIDER_DESCRIPTORS),
|
|
1983
2092
|
);
|
|
@@ -3101,6 +3210,91 @@ export interface SyntheticModelManagerConfig {
|
|
|
3101
3210
|
fetch?: FetchImpl;
|
|
3102
3211
|
}
|
|
3103
3212
|
|
|
3213
|
+
/**
|
|
3214
|
+
* Synthetic's `/openai/v1/models` entry shape (verified live against
|
|
3215
|
+
* api.synthetic.new). It shares no capability field names with the generic
|
|
3216
|
+
* OpenAI-compatible conventions: capabilities arrive in `supported_features`,
|
|
3217
|
+
* modalities in `input_modalities`, the output cap in `max_output_length`, the
|
|
3218
|
+
* accepted `reasoning_effort` vocabulary in `reasoning_parameters.efforts`,
|
|
3219
|
+
* and per-token prices in `pricing` as `$`-prefixed decimal strings.
|
|
3220
|
+
*/
|
|
3221
|
+
interface SyntheticModelRecord extends OpenAICompatibleModelRecord {
|
|
3222
|
+
supported_features?: unknown;
|
|
3223
|
+
input_modalities?: unknown;
|
|
3224
|
+
max_output_length?: unknown;
|
|
3225
|
+
reasoning_parameters?: unknown;
|
|
3226
|
+
pricing?: unknown;
|
|
3227
|
+
}
|
|
3228
|
+
|
|
3229
|
+
/** Synthetic's thinking-off wire tier — a router state, not a user effort. */
|
|
3230
|
+
const SYNTHETIC_WIRE_EFFORT_NONE = "none";
|
|
3231
|
+
/** Output cap for routes that advertise no `max_output_length`. */
|
|
3232
|
+
const SYNTHETIC_FALLBACK_MAX_TOKENS = 8192;
|
|
3233
|
+
|
|
3234
|
+
function toSyntheticStringList(value: unknown): readonly string[] {
|
|
3235
|
+
return Array.isArray(value) ? value.filter((item): item is string => typeof item === "string") : [];
|
|
3236
|
+
}
|
|
3237
|
+
|
|
3238
|
+
/**
|
|
3239
|
+
* Translate Synthetic's per-model `reasoning_effort` vocabulary into an effort
|
|
3240
|
+
* ladder. Every advertised value that names an OMP tier maps verbatim; `none`
|
|
3241
|
+
* is the thinking-off state rather than a tier of its own, so it backs the
|
|
3242
|
+
* `minimal` selector through the wire map (same shape as the Fireworks
|
|
3243
|
+
* `minimal → none` map) and gives these routes a real no-thinking tier.
|
|
3244
|
+
* A route that advertises only `none` (or tiers this client doesn't know)
|
|
3245
|
+
* still gets the minimal-off mapping: falling through to identity inference
|
|
3246
|
+
* would fabricate an unadvertised ladder, and leaving thinking unset would
|
|
3247
|
+
* leak any stale reference ladder past the wire vocabulary.
|
|
3248
|
+
*/
|
|
3249
|
+
function resolveSyntheticThinking(wireEfforts: readonly string[]): ThinkingConfig | undefined {
|
|
3250
|
+
const efforts = THINKING_EFFORTS.filter(effort => wireEfforts.includes(effort));
|
|
3251
|
+
const wireHasNone = wireEfforts.includes(SYNTHETIC_WIRE_EFFORT_NONE);
|
|
3252
|
+
if (efforts.length === 0) {
|
|
3253
|
+
return wireHasNone
|
|
3254
|
+
? { mode: "effort", efforts: [Effort.Minimal], effortMap: { [Effort.Minimal]: SYNTHETIC_WIRE_EFFORT_NONE } }
|
|
3255
|
+
: undefined;
|
|
3256
|
+
}
|
|
3257
|
+
if (!wireHasNone || efforts.includes(Effort.Minimal)) {
|
|
3258
|
+
return { mode: "effort", efforts };
|
|
3259
|
+
}
|
|
3260
|
+
return {
|
|
3261
|
+
mode: "effort",
|
|
3262
|
+
efforts: [Effort.Minimal, ...efforts],
|
|
3263
|
+
effortMap: { [Effort.Minimal]: SYNTHETIC_WIRE_EFFORT_NONE },
|
|
3264
|
+
};
|
|
3265
|
+
}
|
|
3266
|
+
|
|
3267
|
+
/** Synthetic quotes per-token USD as `"$0.000001"`; catalog cost is per-million. */
|
|
3268
|
+
function toSyntheticCostPerMillion(value: unknown): number | undefined {
|
|
3269
|
+
const parsed = toNumber(typeof value === "string" ? value.trim().replace(/^\$/, "") : value);
|
|
3270
|
+
if (parsed === undefined || parsed < 0) {
|
|
3271
|
+
return undefined;
|
|
3272
|
+
}
|
|
3273
|
+
// Scaling a per-token decimal by 1e6 drifts (4.5e-7 → 0.44999999999999996), so
|
|
3274
|
+
// settle on a millionth of a dollar per million tokens — finer than any real tier.
|
|
3275
|
+
return Math.round(parsed * 1e12) / 1e6;
|
|
3276
|
+
}
|
|
3277
|
+
|
|
3278
|
+
function resolveSyntheticCost(
|
|
3279
|
+
pricing: unknown,
|
|
3280
|
+
fallback: ModelSpec<"openai-completions">["cost"],
|
|
3281
|
+
): ModelSpec<"openai-completions">["cost"] {
|
|
3282
|
+
if (!isRecord(pricing)) {
|
|
3283
|
+
return fallback;
|
|
3284
|
+
}
|
|
3285
|
+
const input = toSyntheticCostPerMillion(pricing.prompt);
|
|
3286
|
+
const output = toSyntheticCostPerMillion(pricing.completion);
|
|
3287
|
+
if (input === undefined || output === undefined) {
|
|
3288
|
+
return fallback;
|
|
3289
|
+
}
|
|
3290
|
+
return {
|
|
3291
|
+
input,
|
|
3292
|
+
output,
|
|
3293
|
+
cacheRead: toSyntheticCostPerMillion(pricing.input_cache_reads) ?? fallback.cacheRead,
|
|
3294
|
+
cacheWrite: toSyntheticCostPerMillion(pricing.input_cache_writes) ?? fallback.cacheWrite,
|
|
3295
|
+
};
|
|
3296
|
+
}
|
|
3297
|
+
|
|
3104
3298
|
export function syntheticModelManagerOptions(
|
|
3105
3299
|
config?: SyntheticModelManagerConfig,
|
|
3106
3300
|
): ModelManagerOptions<"openai-completions"> {
|
|
@@ -3124,18 +3318,71 @@ export function syntheticModelManagerOptions(
|
|
|
3124
3318
|
defaults: ModelSpec<"openai-completions">,
|
|
3125
3319
|
_context: OpenAICompatibleModelMapperContext<"openai-completions">,
|
|
3126
3320
|
): ModelSpec<"openai-completions"> => {
|
|
3321
|
+
const record = entry as SyntheticModelRecord;
|
|
3127
3322
|
const reference = references.get(defaults.id);
|
|
3128
3323
|
const referenceSupportsImage = reference?.input.includes("image") ?? false;
|
|
3324
|
+
const features = toSyntheticStringList(record.supported_features);
|
|
3325
|
+
const modalities = toSyntheticStringList(record.input_modalities);
|
|
3326
|
+
const wireEfforts = isRecord(record.reasoning_parameters)
|
|
3327
|
+
? toSyntheticStringList(record.reasoning_parameters.efforts)
|
|
3328
|
+
: [];
|
|
3329
|
+
const wireReasoning = features.includes("reasoning") || wireEfforts.length > 0;
|
|
3330
|
+
const thinking = resolveSyntheticThinking(wireEfforts);
|
|
3331
|
+
// An advertised effort vocabulary is authoritative over the bundled
|
|
3332
|
+
// reference: when the wire names tiers (even only `none`), the
|
|
3333
|
+
// reference's reasoning flag must not re-add a dial the route
|
|
3334
|
+
// doesn't expose. A route with at least one named tier reasons —
|
|
3335
|
+
// even a single tier is a real effort the wire accepts. Only a
|
|
3336
|
+
// vocabulary of `none`/unrecognized values alone is the pure
|
|
3337
|
+
// off-switch: reporting `reasoning: true` there would light up the
|
|
3338
|
+
// effort dial for a dial with one stop. When the wire is silent on
|
|
3339
|
+
// reasoning entirely, the reference gets a vote.
|
|
3340
|
+
const namedTierCount =
|
|
3341
|
+
(thinking?.efforts.length ?? 0) - (wireEfforts.includes(SYNTHETIC_WIRE_EFFORT_NONE) ? 1 : 0);
|
|
3342
|
+
const reasoning =
|
|
3343
|
+
wireReasoning && namedTierCount > 0
|
|
3344
|
+
? true
|
|
3345
|
+
: wireEfforts.length > 0
|
|
3346
|
+
? false
|
|
3347
|
+
: entry.supports_reasoning === true || (reference?.reasoning ?? false);
|
|
3348
|
+
// The router aliases (`syn:*`) and newly added routes carry no
|
|
3349
|
+
// bundled reference, so these advertised capabilities are the only
|
|
3350
|
+
// truth available. Without them such a model lands non-reasoning
|
|
3351
|
+
// (which hides the thinking selector and drops `reasoning_effort`
|
|
3352
|
+
// from every request), text-only, priced at zero, and capped at the
|
|
3353
|
+
// 8k placeholder — a cap low enough that verbose models stop on
|
|
3354
|
+
// `length` each turn and trip recovery compaction.
|
|
3355
|
+
const base = reference ? { ...reference, id: defaults.id, baseUrl } : defaults;
|
|
3129
3356
|
return {
|
|
3130
|
-
...
|
|
3357
|
+
...base,
|
|
3131
3358
|
name: toModelName(entry.name, reference?.name ?? defaults.name),
|
|
3132
|
-
reasoning
|
|
3133
|
-
|
|
3359
|
+
reasoning,
|
|
3360
|
+
...(thinking ? { thinking } : {}),
|
|
3361
|
+
input:
|
|
3362
|
+
modalities.includes("image") || entry.supports_vision === true || referenceSupportsImage
|
|
3363
|
+
? ["text", "image"]
|
|
3364
|
+
: ["text"],
|
|
3365
|
+
// A present `supported_features` list (even empty) is the route's
|
|
3366
|
+
// whole advertised surface: no `tools` entry means no tool
|
|
3367
|
+
// support. The reference still wins when it already vouched for
|
|
3368
|
+
// tools, since a populated wire list can be incomplete; an
|
|
3369
|
+
// explicit reference `false` stays `false` either way.
|
|
3370
|
+
...(record.supported_features !== undefined &&
|
|
3371
|
+
!features.includes("tools") &&
|
|
3372
|
+
reference?.supportsTools !== true
|
|
3373
|
+
? { supportsTools: false }
|
|
3374
|
+
: reference?.supportsTools === false
|
|
3375
|
+
? { supportsTools: false }
|
|
3376
|
+
: {}),
|
|
3377
|
+
cost: resolveSyntheticCost(record.pricing, base.cost),
|
|
3134
3378
|
contextWindow: toPositiveNumber(
|
|
3135
3379
|
entry.context_length,
|
|
3136
3380
|
reference?.contextWindow ?? defaults.contextWindow,
|
|
3137
3381
|
),
|
|
3138
|
-
maxTokens: toPositiveNumber(
|
|
3382
|
+
maxTokens: toPositiveNumber(
|
|
3383
|
+
record.max_output_length ?? entry.max_tokens,
|
|
3384
|
+
reference?.maxTokens ?? SYNTHETIC_FALLBACK_MAX_TOKENS,
|
|
3385
|
+
),
|
|
3139
3386
|
};
|
|
3140
3387
|
},
|
|
3141
3388
|
fetch: config?.fetch,
|
|
@@ -4303,8 +4550,8 @@ export interface GithubCopilotModelManagerConfig {
|
|
|
4303
4550
|
|
|
4304
4551
|
const COPILOT_ANTHROPIC_MODEL_PATTERN = /^claude-(haiku|sonnet|opus|fable|mythos)-\d/;
|
|
4305
4552
|
const isCopilotResponsesModelId = (modelId: string): boolean =>
|
|
4306
|
-
modelId.startsWith("gpt-5") || modelId.startsWith("oswe") || modelId.startsWith("mai-");
|
|
4307
|
-
const COPILOT_CACHE_INVALIDATED_MODEL_IDS = ["mai-code-1-flash-picker"];
|
|
4553
|
+
modelId === "grok-4.5" || modelId.startsWith("gpt-5") || modelId.startsWith("oswe") || modelId.startsWith("mai-");
|
|
4554
|
+
const COPILOT_CACHE_INVALIDATED_MODEL_IDS = ["grok-4.5", "grok-4.5-1m", "mai-code-1-flash-picker"];
|
|
4308
4555
|
|
|
4309
4556
|
function inferCopilotApi(modelId: string): Api {
|
|
4310
4557
|
if (COPILOT_ANTHROPIC_MODEL_PATTERN.test(modelId)) {
|
|
@@ -4657,12 +4904,12 @@ export function anthropicModelManagerOptions(
|
|
|
4657
4904
|
return {
|
|
4658
4905
|
providerId: "anthropic",
|
|
4659
4906
|
modelsDev: {
|
|
4660
|
-
fetch: () =>
|
|
4907
|
+
fetch: () => fetchWellKnownModels(config?.fetch),
|
|
4661
4908
|
map: payload => mapAnthropicModelsDev(payload, baseUrl),
|
|
4662
4909
|
},
|
|
4663
4910
|
...(apiKey && {
|
|
4664
4911
|
fetchDynamicModels: async () => {
|
|
4665
|
-
const modelsDevModels = await
|
|
4912
|
+
const modelsDevModels = await fetchWellKnownModels(config?.fetch)
|
|
4666
4913
|
.then(payload => mapAnthropicModelsDev(payload, baseUrl))
|
|
4667
4914
|
.catch(() => []);
|
|
4668
4915
|
const references = buildAnthropicReferenceMap(modelsDevModels);
|
package/src/types.ts
CHANGED
|
@@ -103,6 +103,8 @@ export interface Usage {
|
|
|
103
103
|
cacheWrite: number;
|
|
104
104
|
/** Sum of input + output + cacheRead + cacheWrite plus provider-side orchestration tokens when reported. */
|
|
105
105
|
totalTokens: number;
|
|
106
|
+
/** Provider-reported occupied context tokens when the value is authoritative but not a billable input/output bucket. */
|
|
107
|
+
contextTokens?: number;
|
|
106
108
|
/** Provider-side orchestration tokens, billed but not part of the conversation prompt/cache buckets. */
|
|
107
109
|
orchestration?: {
|
|
108
110
|
/** Non-cached orchestration input tokens. */
|