@oh-my-pi/pi-catalog 17.3.3 → 17.3.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/dist/types/compat/openai.d.ts +2 -0
- package/dist/types/hosts.d.ts +1 -1
- package/dist/types/identity/family.d.ts +23 -0
- package/dist/types/openai-pricing.d.ts +17 -0
- package/dist/types/provider-models/descriptors.d.ts +4 -4
- package/dist/types/provider-models/openai-compat.d.ts +17 -1
- package/dist/types/types.d.ts +15 -1
- package/dist/types/wire/codex.d.ts +3 -0
- package/dist/types/wire/github-copilot.d.ts +8 -0
- package/package.json +4 -4
- package/src/compat/openai.ts +54 -11
- package/src/discovery/codex.ts +3 -1
- package/src/hosts.ts +1 -1
- package/src/identity/family.ts +47 -1
- package/src/model-thinking.ts +30 -5
- package/src/models.json +2834 -810
- package/src/openai-pricing.ts +29 -0
- package/src/provider-models/cache-provider-id.ts +14 -0
- package/src/provider-models/descriptors.ts +3 -3
- package/src/provider-models/openai-compat.ts +117 -53
- package/src/types.ts +15 -0
- package/src/wire/codex.ts +3 -0
- package/src/wire/github-copilot.ts +32 -0
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
import type { TokenCost } from "./types";
|
|
2
|
+
|
|
3
|
+
/** Standard GPT-5.6 Sol rates used by the Daybreak Blue aliases. */
|
|
4
|
+
export const OPENAI_GPT_56_SOL_STANDARD_COST = {
|
|
5
|
+
input: 5,
|
|
6
|
+
output: 30,
|
|
7
|
+
cacheRead: 0.5,
|
|
8
|
+
cacheWrite: 6.25,
|
|
9
|
+
} as const satisfies TokenCost;
|
|
10
|
+
|
|
11
|
+
/** Standard GPT-5.6 Cyber rates used by the Daybreak Red aliases. */
|
|
12
|
+
export const OPENAI_GPT_56_CYBER_STANDARD_COST = {
|
|
13
|
+
input: 12.5,
|
|
14
|
+
output: 75,
|
|
15
|
+
cacheRead: 1.25,
|
|
16
|
+
cacheWrite: 15.625,
|
|
17
|
+
} as const satisfies TokenCost;
|
|
18
|
+
|
|
19
|
+
/** Resolve standard rates for Codex-prefixed Daybreak aliases. */
|
|
20
|
+
export function resolveOpenAIDaybreakStandardCost(modelId: string): TokenCost | undefined {
|
|
21
|
+
switch (modelId) {
|
|
22
|
+
case "gpt-daybreak-blue-latest":
|
|
23
|
+
return OPENAI_GPT_56_SOL_STANDARD_COST;
|
|
24
|
+
case "gpt-daybreak-red-latest":
|
|
25
|
+
return OPENAI_GPT_56_CYBER_STANDARD_COST;
|
|
26
|
+
default:
|
|
27
|
+
return undefined;
|
|
28
|
+
}
|
|
29
|
+
}
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { PERSONAL_GITHUB_COPILOT_BASE_URL } from "../wire/github-copilot";
|
|
2
|
+
|
|
1
3
|
export interface ModelCacheProviderIdOptions {
|
|
2
4
|
apiKey?: string;
|
|
3
5
|
baseUrl?: string;
|
|
@@ -56,6 +58,18 @@ export function resolveModelCacheProviderId(providerId: string, options: ModelCa
|
|
|
56
58
|
const scope = `${options.apiKey ?? ""}\u0000${discoveryBaseUrl}`;
|
|
57
59
|
return `${providerId}:models-v1:${Bun.hash(scope).toString(36)}`;
|
|
58
60
|
}
|
|
61
|
+
case "github-copilot": {
|
|
62
|
+
// Copilot model specs bake in the plan-specific endpoint (personal vs
|
|
63
|
+
// Business/Enterprise) resolved from the credential. Discovery writes an
|
|
64
|
+
// authoritative cache, so `online-if-uncached` serves it for the full
|
|
65
|
+
// TTL without re-probing. Keying the namespace on the credential means
|
|
66
|
+
// switching `COPILOT_GITHUB_TOKEN` to a different account misses the
|
|
67
|
+
// prior endpoint's cache and re-runs discovery instead of hitting the
|
|
68
|
+
// stale host and 403ing (PR #8510 review).
|
|
69
|
+
const baseUrl = options.baseUrl ?? PERSONAL_GITHUB_COPILOT_BASE_URL;
|
|
70
|
+
const scope = `${options.apiKey ?? ""}\u0000${baseUrl}`;
|
|
71
|
+
return `github-copilot:models-v1:${Bun.hash(scope).toString(36)}`;
|
|
72
|
+
}
|
|
59
73
|
case "openrouter":
|
|
60
74
|
return "openrouter:pseudo-api";
|
|
61
75
|
case "vllm": {
|
|
@@ -478,13 +478,13 @@ export const CATALOG_PROVIDERS = [
|
|
|
478
478
|
},
|
|
479
479
|
{
|
|
480
480
|
id: "xai",
|
|
481
|
-
defaultModel: "grok-4
|
|
481
|
+
defaultModel: "grok-4.5",
|
|
482
482
|
envVars: ["XAI_API_KEY"],
|
|
483
483
|
createModelManagerOptions: (config: ModelManagerConfig) => xaiModelManagerOptions(config),
|
|
484
484
|
},
|
|
485
485
|
{
|
|
486
486
|
id: "xai-oauth",
|
|
487
|
-
defaultModel: "grok-4.
|
|
487
|
+
defaultModel: "grok-4.5",
|
|
488
488
|
envVars: ["XAI_OAUTH_TOKEN", "XAI_API_KEY"],
|
|
489
489
|
createModelManagerOptions: (config: ModelManagerConfig) => xaiOAuthModelManagerOptions(config),
|
|
490
490
|
catalogDiscovery: {
|
|
@@ -522,7 +522,7 @@ export const CATALOG_PROVIDERS = [
|
|
|
522
522
|
},
|
|
523
523
|
{
|
|
524
524
|
id: "zai",
|
|
525
|
-
defaultModel: "glm-5.
|
|
525
|
+
defaultModel: "glm-5.3",
|
|
526
526
|
envVars: ["ZAI_API_KEY"],
|
|
527
527
|
createModelManagerOptions: (config: ModelManagerConfig) => zaiModelManagerOptions(config),
|
|
528
528
|
catalogDiscovery: { label: "zAI" },
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
import { USER_AGENT } from "@oh-my-pi/pi-utils";
|
|
2
2
|
import * as logger from "@oh-my-pi/pi-utils/logger";
|
|
3
|
+
import { xaiResponsesReasoningEffortMap } from "../compat/openai";
|
|
3
4
|
import {
|
|
5
|
+
DEFAULT_OPENAI_COMPATIBLE_DISCOVERY_TIMEOUT_MS,
|
|
4
6
|
fetchOpenAICompatibleModels,
|
|
5
7
|
type OpenAICompatibleModelMapperContext,
|
|
6
8
|
type OpenAICompatibleModelRecord,
|
|
@@ -19,12 +21,23 @@ import {
|
|
|
19
21
|
import { resolveModelReference } from "../identity/reference";
|
|
20
22
|
import type { ModelManagerOptions } from "../model-manager";
|
|
21
23
|
import { type GeneratedProvider, getBundledModels } from "../models";
|
|
22
|
-
import
|
|
24
|
+
import { OPENAI_GPT_56_CYBER_STANDARD_COST, OPENAI_GPT_56_SOL_STANDARD_COST } from "../openai-pricing";
|
|
25
|
+
import type {
|
|
26
|
+
Api,
|
|
27
|
+
FetchImpl,
|
|
28
|
+
LongContextTokenCost,
|
|
29
|
+
Model,
|
|
30
|
+
ModelSpec,
|
|
31
|
+
OpenAICompat,
|
|
32
|
+
Provider,
|
|
33
|
+
ThinkingConfig,
|
|
34
|
+
} from "../types";
|
|
23
35
|
import { discoveryFetch, isAnthropicOAuthToken, isRecord, toBoolean, toNumber, toPositiveNumber } from "../utils";
|
|
24
36
|
import { ALIBABA_TOKEN_PLAN_BASE_URL, parseAlibabaTokenPlanCredential } from "../wire/alibaba-token-plan";
|
|
25
37
|
import { coreWeaveProjectHeaders } from "../wire/coreweave";
|
|
26
38
|
import {
|
|
27
39
|
COPILOT_API_HEADERS,
|
|
40
|
+
discoverGitHubCopilotApiEndpoint,
|
|
28
41
|
getGitHubCopilotBaseUrl,
|
|
29
42
|
isPersonalGitHubCopilotBaseUrl,
|
|
30
43
|
parseGitHubCopilotApiKey,
|
|
@@ -867,6 +880,7 @@ export function umansModelManagerOptions(config?: UmansModelManagerConfig): Mode
|
|
|
867
880
|
// ---------------------------------------------------------------------------
|
|
868
881
|
|
|
869
882
|
const OPENAI_API_BASE_URL = "https://api.openai.com/v1";
|
|
883
|
+
/** GPT-5.6 rates applied when a first-party request exceeds 272K input tokens. */
|
|
870
884
|
export const OPENAI_GPT_56_LONG_CONTEXT_COSTS = {
|
|
871
885
|
luna: {
|
|
872
886
|
inputThreshold: 272_000,
|
|
@@ -889,20 +903,7 @@ export const OPENAI_GPT_56_LONG_CONTEXT_COSTS = {
|
|
|
889
903
|
cacheRead: 0.4,
|
|
890
904
|
cacheWrite: 5,
|
|
891
905
|
},
|
|
892
|
-
} as const
|
|
893
|
-
const OPENAI_GPT_56_SOL_STANDARD_COST = {
|
|
894
|
-
input: 5,
|
|
895
|
-
output: 30,
|
|
896
|
-
cacheRead: 0.5,
|
|
897
|
-
cacheWrite: 6.25,
|
|
898
|
-
longContext: OPENAI_GPT_56_LONG_CONTEXT_COSTS.sol,
|
|
899
|
-
} as const;
|
|
900
|
-
const OPENAI_GPT_56_CYBER_STANDARD_COST = {
|
|
901
|
-
input: 12.5,
|
|
902
|
-
output: 75,
|
|
903
|
-
cacheRead: 1.25,
|
|
904
|
-
cacheWrite: 15.625,
|
|
905
|
-
} as const;
|
|
906
|
+
} as const satisfies Readonly<Record<"luna" | "sol" | "terra", LongContextTokenCost>>;
|
|
906
907
|
|
|
907
908
|
export interface OpenAIModelManagerConfig {
|
|
908
909
|
apiKey?: string;
|
|
@@ -936,7 +937,10 @@ export const OPENAI_DAYBREAK_CURATED_FALLBACK_MODELS: readonly ModelSpec<"openai
|
|
|
936
937
|
baseUrl: OPENAI_API_BASE_URL,
|
|
937
938
|
reasoning: true,
|
|
938
939
|
input: ["text", "image"],
|
|
939
|
-
cost:
|
|
940
|
+
cost: {
|
|
941
|
+
...OPENAI_GPT_56_SOL_STANDARD_COST,
|
|
942
|
+
longContext: OPENAI_GPT_56_LONG_CONTEXT_COSTS.sol,
|
|
943
|
+
},
|
|
940
944
|
contextWindow: 1_050_000,
|
|
941
945
|
maxTokens: 128_000,
|
|
942
946
|
},
|
|
@@ -1255,8 +1259,23 @@ export interface XaiModelManagerConfig {
|
|
|
1255
1259
|
fetch?: FetchImpl;
|
|
1256
1260
|
}
|
|
1257
1261
|
|
|
1258
|
-
export function xaiModelManagerOptions(config?: XaiModelManagerConfig): ModelManagerOptions<"openai-
|
|
1259
|
-
return
|
|
1262
|
+
export function xaiModelManagerOptions(config?: XaiModelManagerConfig): ModelManagerOptions<"openai-responses"> {
|
|
1263
|
+
return {
|
|
1264
|
+
...createOpenAICompatibleModelManagerOptions({
|
|
1265
|
+
api: "openai-responses",
|
|
1266
|
+
providerId: "xai",
|
|
1267
|
+
defaultBaseUrl: "https://api.x.ai/v1",
|
|
1268
|
+
config,
|
|
1269
|
+
requireApiKey: true,
|
|
1270
|
+
mapModel: mapWithBundledReference,
|
|
1271
|
+
}),
|
|
1272
|
+
// Completions → Responses migration: a fresh authoritative cache written
|
|
1273
|
+
// by the old resolver stores `api: "openai-completions"` for these ids.
|
|
1274
|
+
// Without a drop list, `online-if-uncached` skips the network and
|
|
1275
|
+
// `mergeDynamicModel` lets the cached api win over the new static
|
|
1276
|
+
// Responses entries until TTL expiry.
|
|
1277
|
+
dropCachedModelIdsOnStaticMismatch: getBundledModels("xai").map(model => model.id),
|
|
1278
|
+
};
|
|
1260
1279
|
}
|
|
1261
1280
|
|
|
1262
1281
|
export interface XaiOAuthModelManagerConfig {
|
|
@@ -1314,6 +1333,7 @@ export const XAI_OAUTH_CURATED_MODELS: readonly XAICuratedModel[] = [
|
|
|
1314
1333
|
},
|
|
1315
1334
|
{ id: "grok-4.3", contextWindow: 1_000_000, name: "Grok 4.3", input: ["text", "image"] },
|
|
1316
1335
|
{ id: "grok-4.5", contextWindow: 500_000, name: "Grok 4.5", input: ["text", "image"] },
|
|
1336
|
+
{ id: "grok-4.6", contextWindow: 500_000, name: "Grok 4.6", input: ["text", "image"] },
|
|
1317
1337
|
// grok-4.20-multi-agent-0309 is text-only per the bundled catalog; omit `input` for the default.
|
|
1318
1338
|
{ id: "grok-4.20-multi-agent-0309", contextWindow: 2_000_000, name: "Grok 4.20 (Multi-Agent)" },
|
|
1319
1339
|
{
|
|
@@ -1350,21 +1370,47 @@ const XAI_NON_CHAT_PREFIXES = ["grok-imagine-", "grok-stt-", "grok-voice-"] as c
|
|
|
1350
1370
|
function withXaiOAuthCompatDefaults(model: ModelSpec<"openai-responses">): ModelSpec<"openai-responses"> {
|
|
1351
1371
|
const compat = {
|
|
1352
1372
|
...(model.compat ?? {}),
|
|
1353
|
-
includeEncryptedReasoning: model.compat?.includeEncryptedReasoning ??
|
|
1354
|
-
filterReasoningHistory: model.compat?.filterReasoningHistory ??
|
|
1373
|
+
includeEncryptedReasoning: model.compat?.includeEncryptedReasoning ?? true,
|
|
1374
|
+
filterReasoningHistory: model.compat?.filterReasoningHistory ?? false,
|
|
1355
1375
|
supportsImageDetailOriginal: model.compat?.supportsImageDetailOriginal ?? false,
|
|
1356
1376
|
omitReasoningEffort: model.compat?.omitReasoningEffort ?? !isGrokReasoningEffortCapable(model.id),
|
|
1357
1377
|
};
|
|
1358
1378
|
return { ...model, compat };
|
|
1359
1379
|
}
|
|
1360
1380
|
|
|
1361
|
-
// Hermes-agent parity
|
|
1362
|
-
//
|
|
1363
|
-
//
|
|
1364
|
-
//
|
|
1365
|
-
|
|
1366
|
-
|
|
1367
|
-
|
|
1381
|
+
// Hermes-agent parity for `minimal -> low` (see hermes-agent/agent/transports/
|
|
1382
|
+
// codex.py:92). Multi-agent Grok keeps `xhigh` unmapped (agent-count mode);
|
|
1383
|
+
// other first-party SKUs clamp leftover `xhigh`/`max` to `high`.
|
|
1384
|
+
// `resolveModelThinking` folds this into `model.thinking.effortMap`.
|
|
1385
|
+
|
|
1386
|
+
/**
|
|
1387
|
+
* Bake first-party xAI Responses effort-dial metadata onto a catalog spec.
|
|
1388
|
+
*
|
|
1389
|
+
* models.dev marks many Grok SKUs as reasoners and the thinking rebake would
|
|
1390
|
+
* otherwise emit a default `minimal/low/medium/high` dial. api.x.ai only
|
|
1391
|
+
* accepts `reasoning.effort` for {@link isGrokReasoningEffortCapable} ids —
|
|
1392
|
+
* off-allowlist reasoners (`grok-code-fast-1`, `grok-build-0.1`,
|
|
1393
|
+
* `grok-4.20-0309-reasoning`, …) 400 if the param is sent. SuperGrok
|
|
1394
|
+
* (`xai-oauth`) already curates this via {@link mergeCuratedIntoModel}; paid
|
|
1395
|
+
* `xai` rows come from stencil.so and need the same wire facts in the exported
|
|
1396
|
+
* `models.json` so direct catalog readers do not present an unsupported dial.
|
|
1397
|
+
*
|
|
1398
|
+
* Explicit `compat.supportsReasoningEffort` / `omitReasoningEffort` win.
|
|
1399
|
+
*/
|
|
1400
|
+
export function applyXaiResponsesThinkingPolicy(model: ModelSpec<"openai-responses">): ModelSpec<"openai-responses"> {
|
|
1401
|
+
const effortCapable = model.compat?.supportsReasoningEffort ?? isGrokReasoningEffortCapable(model.id);
|
|
1402
|
+
const compat = {
|
|
1403
|
+
...(model.compat ?? {}),
|
|
1404
|
+
supportsReasoningEffort: effortCapable,
|
|
1405
|
+
omitReasoningEffort: model.compat?.omitReasoningEffort ?? !effortCapable,
|
|
1406
|
+
};
|
|
1407
|
+
if (effortCapable) {
|
|
1408
|
+
compat.reasoningEffortMap = { ...xaiResponsesReasoningEffortMap(model.id) };
|
|
1409
|
+
} else {
|
|
1410
|
+
delete compat.reasoningEffortMap;
|
|
1411
|
+
}
|
|
1412
|
+
return { ...model, compat };
|
|
1413
|
+
}
|
|
1368
1414
|
|
|
1369
1415
|
// xai-oauth's /v1/models exposes no per-request output limit on the OAuth
|
|
1370
1416
|
// (Grok Build / SuperGrok) surface, so the curated catalog owns `maxTokens`
|
|
@@ -1379,9 +1425,9 @@ const XAI_REASONING_EFFORT_MAP = { minimal: "low" } as const;
|
|
|
1379
1425
|
// reasoning metadata and fetchOpenAICompatibleModels defaults reasoning to
|
|
1380
1426
|
// false). Caller supplies a `base` Model (either a freshly synthesised seed
|
|
1381
1427
|
// or a dynamic-fetched entry); the helper layers curated fields on top.
|
|
1382
|
-
// The
|
|
1383
|
-
//
|
|
1384
|
-
//
|
|
1428
|
+
// The effort remap from {@link xaiResponsesReasoningEffortMap} is merged
|
|
1429
|
+
// only onto effort-capable rows. Off-allowlist reasoners omit the wire
|
|
1430
|
+
// param, so a map on those specs is dead weight.
|
|
1385
1431
|
// The effort-dial pair (`supportsReasoningEffort`/`omitReasoningEffort`) is
|
|
1386
1432
|
// authoritative: a stale flag on `base` (previous snapshot or dynamic fetch)
|
|
1387
1433
|
// must not outlive an allowlist change in identity/family.ts.
|
|
@@ -1392,13 +1438,17 @@ function mergeCuratedIntoModel(
|
|
|
1392
1438
|
const effortCapable = curated.supportsReasoningEffort ?? isGrokReasoningEffortCapable(curated.id);
|
|
1393
1439
|
const compat = {
|
|
1394
1440
|
...(base.compat ?? {}),
|
|
1395
|
-
|
|
1396
|
-
|
|
1397
|
-
filterReasoningHistory: base.compat?.filterReasoningHistory ?? true,
|
|
1441
|
+
includeEncryptedReasoning: base.compat?.includeEncryptedReasoning ?? true,
|
|
1442
|
+
filterReasoningHistory: false,
|
|
1398
1443
|
supportsImageDetailOriginal: base.compat?.supportsImageDetailOriginal ?? false,
|
|
1399
1444
|
omitReasoningEffort: !effortCapable,
|
|
1400
1445
|
supportsReasoningEffort: effortCapable,
|
|
1401
1446
|
};
|
|
1447
|
+
if (effortCapable) {
|
|
1448
|
+
compat.reasoningEffortMap = { ...xaiResponsesReasoningEffortMap(curated.id) };
|
|
1449
|
+
} else {
|
|
1450
|
+
delete compat.reasoningEffortMap;
|
|
1451
|
+
}
|
|
1402
1452
|
return {
|
|
1403
1453
|
...base,
|
|
1404
1454
|
contextWindow: curated.contextWindow,
|
|
@@ -1500,7 +1550,7 @@ export function buildXaiOAuthStaticSeed(baseUrl?: string): ModelSpec<"openai-res
|
|
|
1500
1550
|
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
|
1501
1551
|
contextWindow: curated.contextWindow,
|
|
1502
1552
|
maxTokens: curated.contextWindow,
|
|
1503
|
-
compat: { reasoningEffortMap:
|
|
1553
|
+
compat: { reasoningEffortMap: xaiResponsesReasoningEffortMap(curated.id) },
|
|
1504
1554
|
};
|
|
1505
1555
|
return mergeCuratedIntoModel(base, curated);
|
|
1506
1556
|
});
|
|
@@ -3632,15 +3682,19 @@ export function basetenModelManagerOptions(
|
|
|
3632
3682
|
const features = Array.isArray(raw.supported_features) ? raw.supported_features : [];
|
|
3633
3683
|
const modalities = Array.isArray(raw.input_modalities) ? raw.input_modalities : [];
|
|
3634
3684
|
|
|
3635
|
-
// Baseten's
|
|
3636
|
-
//
|
|
3637
|
-
|
|
3685
|
+
// Baseten's discovery flags are not enough to enable OMP reasoning for every
|
|
3686
|
+
// model. Only models with a verified Baseten reasoning policy are enabled
|
|
3687
|
+
// here; an unknown model may use a different reasoning wire shape or effort
|
|
3688
|
+
// vocabulary, which OMP must not guess.
|
|
3689
|
+
const isSupportedBasetenReasoningModel =
|
|
3690
|
+
isKimiK3ModelId(defaults.id) ||
|
|
3638
3691
|
defaults.id === "openai/gpt-oss-120b" ||
|
|
3692
|
+
defaults.id === "deepseek-ai/DeepSeek-V4-Pro" ||
|
|
3639
3693
|
defaults.id === "zai-org/GLM-5.2" ||
|
|
3640
3694
|
defaults.id === "zai-org/GLM-5.2-Fast";
|
|
3641
|
-
const isBasetenNativeReasoning = isEffortReasoning || defaults.id === "deepseek-ai/DeepSeek-V4-Pro";
|
|
3642
3695
|
const reasoning =
|
|
3643
|
-
|
|
3696
|
+
isSupportedBasetenReasoningModel &&
|
|
3697
|
+
(features.includes("reasoning") || features.includes("reasoning_effort"));
|
|
3644
3698
|
const supportsTools = features.includes("tools") ? undefined : false;
|
|
3645
3699
|
const vision = modalities.includes("image") || (reference?.input.includes("image") ?? false);
|
|
3646
3700
|
|
|
@@ -3654,14 +3708,7 @@ export function basetenModelManagerOptions(
|
|
|
3654
3708
|
|
|
3655
3709
|
const contextWindow = toPositiveNumber(raw.context_length, reference?.contextWindow ?? defaults.contextWindow);
|
|
3656
3710
|
const maxTokens = toPositiveNumber(raw.max_completion_tokens, reference?.maxTokens ?? defaults.maxTokens);
|
|
3657
|
-
|
|
3658
3711
|
const baseModel = mapWithBundledReference(entry, defaults, reference);
|
|
3659
|
-
const thinking = isEffortReasoning
|
|
3660
|
-
? {
|
|
3661
|
-
mode: "effort" as const,
|
|
3662
|
-
efforts: [Effort.High, Effort.Max],
|
|
3663
|
-
}
|
|
3664
|
-
: undefined;
|
|
3665
3712
|
|
|
3666
3713
|
return {
|
|
3667
3714
|
...baseModel,
|
|
@@ -3670,7 +3717,6 @@ export function basetenModelManagerOptions(
|
|
|
3670
3717
|
cost,
|
|
3671
3718
|
contextWindow,
|
|
3672
3719
|
maxTokens,
|
|
3673
|
-
...(thinking ? { thinking } : {}),
|
|
3674
3720
|
...(supportsTools === false ? { supportsTools } : {}),
|
|
3675
3721
|
};
|
|
3676
3722
|
},
|
|
@@ -5191,6 +5237,7 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
|
|
|
5191
5237
|
const resolveReference = createReferenceResolver(getProviderReferences);
|
|
5192
5238
|
return {
|
|
5193
5239
|
providerId: "github-copilot",
|
|
5240
|
+
cacheProviderId: resolveModelCacheProviderId("github-copilot", { apiKey: rawApiKey, baseUrl }),
|
|
5194
5241
|
dropCachedModelIdsOnStaticMismatch: COPILOT_CACHE_INVALIDATED_MODEL_IDS,
|
|
5195
5242
|
// COPILOT_API_HEADERS are compile-time constants (User-Agent + API
|
|
5196
5243
|
// version), not credentials. The cache omits all request headers for
|
|
@@ -5201,11 +5248,17 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
|
|
|
5201
5248
|
restorableHeaderFallback: { ...COPILOT_API_HEADERS },
|
|
5202
5249
|
...(apiKey && {
|
|
5203
5250
|
fetchDynamicModels: async () => {
|
|
5251
|
+
const fetchImpl = discoveryFetch(config?.fetch);
|
|
5252
|
+
const requestBaseUrl = isPersonalGitHubCopilotBaseUrl(baseUrl)
|
|
5253
|
+
? ((await withCatalogDiscoveryTimeout(DEFAULT_OPENAI_COMPATIBLE_DISCOVERY_TIMEOUT_MS, signal =>
|
|
5254
|
+
discoverGitHubCopilotApiEndpoint(apiKey, fetchImpl, signal),
|
|
5255
|
+
)) ?? baseUrl)
|
|
5256
|
+
: baseUrl;
|
|
5204
5257
|
const longContextVariants: ModelSpec<Api>[] = [];
|
|
5205
5258
|
const models = await fetchOpenAICompatibleModels<Api>({
|
|
5206
5259
|
api: "openai-completions",
|
|
5207
5260
|
provider: "github-copilot",
|
|
5208
|
-
baseUrl,
|
|
5261
|
+
baseUrl: requestBaseUrl,
|
|
5209
5262
|
apiKey,
|
|
5210
5263
|
headers: COPILOT_API_HEADERS,
|
|
5211
5264
|
mapModel: (
|
|
@@ -5251,7 +5304,7 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
|
|
|
5251
5304
|
const input: ModelSpec<Api>["input"] =
|
|
5252
5305
|
supportsVision === true
|
|
5253
5306
|
? ["text", "image"]
|
|
5254
|
-
: supportsVision === false || !isPersonalGitHubCopilotBaseUrl(
|
|
5307
|
+
: supportsVision === false || !isPersonalGitHubCopilotBaseUrl(requestBaseUrl)
|
|
5255
5308
|
? ["text"]
|
|
5256
5309
|
: (reference?.input ?? defaults.input);
|
|
5257
5310
|
// With COPILOT_API_HEADERS the served window is the long-context
|
|
@@ -5272,7 +5325,7 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
|
|
|
5272
5325
|
...reference,
|
|
5273
5326
|
api,
|
|
5274
5327
|
provider: "github-copilot",
|
|
5275
|
-
baseUrl,
|
|
5328
|
+
baseUrl: requestBaseUrl,
|
|
5276
5329
|
name,
|
|
5277
5330
|
input,
|
|
5278
5331
|
contextWindow: defaultTierWindow,
|
|
@@ -5294,7 +5347,7 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
|
|
|
5294
5347
|
: {
|
|
5295
5348
|
...defaults,
|
|
5296
5349
|
api,
|
|
5297
|
-
baseUrl,
|
|
5350
|
+
baseUrl: requestBaseUrl,
|
|
5298
5351
|
name,
|
|
5299
5352
|
input,
|
|
5300
5353
|
contextWindow: defaultTierWindow,
|
|
@@ -5340,7 +5393,7 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
|
|
|
5340
5393
|
}
|
|
5341
5394
|
return base;
|
|
5342
5395
|
},
|
|
5343
|
-
fetch:
|
|
5396
|
+
fetch: fetchImpl,
|
|
5344
5397
|
});
|
|
5345
5398
|
if (models === null) {
|
|
5346
5399
|
return null;
|
|
@@ -5704,6 +5757,15 @@ function openAiCompletionsDescriptor(
|
|
|
5704
5757
|
return simpleModelsDevDescriptor(modelsDevKey, providerId, "openai-completions", baseUrl, options);
|
|
5705
5758
|
}
|
|
5706
5759
|
|
|
5760
|
+
function openAiResponsesDescriptor(
|
|
5761
|
+
modelsDevKey: string,
|
|
5762
|
+
providerId: string,
|
|
5763
|
+
baseUrl: string,
|
|
5764
|
+
options: Omit<ModelsDevProviderDescriptor, "modelsDevKey" | "providerId" | "api" | "baseUrl"> = {},
|
|
5765
|
+
): ModelsDevProviderDescriptor {
|
|
5766
|
+
return simpleModelsDevDescriptor(modelsDevKey, providerId, "openai-responses", baseUrl, options);
|
|
5767
|
+
}
|
|
5768
|
+
|
|
5707
5769
|
function anthropicMessagesDescriptor(
|
|
5708
5770
|
modelsDevKey: string,
|
|
5709
5771
|
providerId: string,
|
|
@@ -5828,7 +5890,9 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
|
|
|
5828
5890
|
defaultContextWindow: 131072,
|
|
5829
5891
|
}),
|
|
5830
5892
|
// --- xAI ---
|
|
5831
|
-
|
|
5893
|
+
openAiResponsesDescriptor("xai", "xai", "https://api.x.ai/v1", {
|
|
5894
|
+
transformModel: model => applyXaiResponsesThinkingPolicy(model as ModelSpec<"openai-responses">),
|
|
5895
|
+
}),
|
|
5832
5896
|
// --- DeepSeek ---
|
|
5833
5897
|
openAiCompletionsDescriptor("deepseek", "deepseek", "https://api.deepseek.com", {
|
|
5834
5898
|
// Only ship the v4 family as built-ins; older deepseek-chat / deepseek-reasoner
|
package/src/types.ts
CHANGED
|
@@ -363,6 +363,13 @@ export interface OpenAICompat {
|
|
|
363
363
|
* model id. Default: true. Issue #5606.
|
|
364
364
|
*/
|
|
365
365
|
supportsSamplingParams?: boolean;
|
|
366
|
+
/**
|
|
367
|
+
* Whether presence/frequency penalties and stop sequences may be sent.
|
|
368
|
+
* First-party xAI `/v1/responses` rejects penalty fields for every model.
|
|
369
|
+
* xAI reasoning models also reject them (and `stop`) on chat completions.
|
|
370
|
+
* When unset, auto-detected. Default: true.
|
|
371
|
+
*/
|
|
372
|
+
supportsPenaltyAndStopParams?: boolean;
|
|
366
373
|
/** Always send a max-token field when the caller did not provide one. Default: auto-detected (Kimi-family models derive TPM limits from max_tokens). */
|
|
367
374
|
alwaysSendMaxTokens?: boolean;
|
|
368
375
|
/** Whether Responses-API tool-call/result history must be strictly paired. Default: auto-detected (Azure OpenAI, GitHub Copilot). */
|
|
@@ -578,6 +585,7 @@ export interface ResolvedOpenAISharedCompat {
|
|
|
578
585
|
reasoningEffortMap: Partial<Record<Effort, string>>;
|
|
579
586
|
supportsReasoningParams: boolean;
|
|
580
587
|
supportsSamplingParams: boolean;
|
|
588
|
+
supportsPenaltyAndStopParams: boolean;
|
|
581
589
|
thinkingFormat: OpenAIReasoningFormat;
|
|
582
590
|
/** Kimi Code transport selected by live per-model protocol metadata. */
|
|
583
591
|
kimiApiFormat?: OpenAICompat["kimiApiFormat"];
|
|
@@ -643,6 +651,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
|
|
|
643
651
|
| "reasoningEffortMap"
|
|
644
652
|
| "supportsReasoningParams"
|
|
645
653
|
| "supportsSamplingParams"
|
|
654
|
+
| "supportsPenaltyAndStopParams"
|
|
646
655
|
| "thinkingFormat"
|
|
647
656
|
| "kimiApiFormat"
|
|
648
657
|
| "reasoningDisableMode"
|
|
@@ -711,6 +720,12 @@ export interface ResolvedOpenAIResponsesCompat extends ResolvedOpenAISharedCompa
|
|
|
711
720
|
strictResponsesPairing: boolean;
|
|
712
721
|
supportsImageDetailOriginal: boolean;
|
|
713
722
|
supportsObfuscationOptOut: boolean;
|
|
723
|
+
/**
|
|
724
|
+
* Whether `reasoning.summary` may be sent. First-party xAI `/v1/responses`
|
|
725
|
+
* rejects the field; handlers pass `null` so the wire omits it instead of
|
|
726
|
+
* filling `"auto"`.
|
|
727
|
+
*/
|
|
728
|
+
supportsReasoningSummary: boolean;
|
|
714
729
|
streamIdleTimeoutMs?: number;
|
|
715
730
|
vercelGatewayRouting?: OpenAICompat["vercelGatewayRouting"];
|
|
716
731
|
/** The model sits behind Vercel AI Gateway's Responses endpoint. */
|
package/src/wire/codex.ts
CHANGED
|
@@ -11,6 +11,8 @@ export const CODEX_CLIENT_VERSION = "0.144.1";
|
|
|
11
11
|
|
|
12
12
|
export const OPENAI_HEADERS = {
|
|
13
13
|
BETA: "OpenAI-Beta",
|
|
14
|
+
/** Codex feature-negotiation header; values identify opt-in wire protocols. */
|
|
15
|
+
CODEX_BETA_FEATURES: "x-codex-beta-features",
|
|
14
16
|
ACCOUNT_ID: "chatgpt-account-id",
|
|
15
17
|
ORIGINATOR: "originator",
|
|
16
18
|
VERSION: "version",
|
|
@@ -32,6 +34,7 @@ export const OPENAI_HEADERS = {
|
|
|
32
34
|
export const OPENAI_HEADER_VALUES = {
|
|
33
35
|
BETA_RESPONSES: "responses=experimental",
|
|
34
36
|
BETA_RESPONSES_WEBSOCKETS_V2: "responses_websockets=2026-02-06",
|
|
37
|
+
REMOTE_COMPACTION_V2: "remote_compaction_v2",
|
|
35
38
|
ORIGINATOR_CODEX: "pi",
|
|
36
39
|
} as const;
|
|
37
40
|
|
|
@@ -1,3 +1,6 @@
|
|
|
1
|
+
import type { FetchImpl } from "../types";
|
|
2
|
+
import { isRecord } from "../utils";
|
|
3
|
+
|
|
1
4
|
/**
|
|
2
5
|
* GitHub Copilot wire metadata: API-key envelope parsing and endpoint
|
|
3
6
|
* derivation shared by catalog discovery and the pi-ai OAuth flow. The device
|
|
@@ -72,6 +75,35 @@ export function normalizeGitHubCopilotApiEndpoint(input: string | undefined): st
|
|
|
72
75
|
return undefined;
|
|
73
76
|
}
|
|
74
77
|
}
|
|
78
|
+
/**
|
|
79
|
+
* Resolve the plan-specific Copilot API endpoint advertised for a GitHub token.
|
|
80
|
+
* Login and raw environment-token discovery share this best-effort probe. Pass
|
|
81
|
+
* a `signal` to bound it against the same discovery deadline as `/models`; a
|
|
82
|
+
* stalled probe otherwise blocks discovery indefinitely.
|
|
83
|
+
*/
|
|
84
|
+
export async function discoverGitHubCopilotApiEndpoint(
|
|
85
|
+
token: string,
|
|
86
|
+
fetchImpl: FetchImpl,
|
|
87
|
+
signal?: AbortSignal,
|
|
88
|
+
): Promise<string | undefined> {
|
|
89
|
+
try {
|
|
90
|
+
const response = await fetchImpl("https://api.github.com/copilot_internal/user", {
|
|
91
|
+
headers: {
|
|
92
|
+
Accept: "application/json",
|
|
93
|
+
Authorization: `token ${token}`,
|
|
94
|
+
...OPENCODE_HEADERS,
|
|
95
|
+
},
|
|
96
|
+
signal,
|
|
97
|
+
});
|
|
98
|
+
if (!response.ok) return undefined;
|
|
99
|
+
const data: unknown = await response.json();
|
|
100
|
+
if (!isRecord(data) || !isRecord(data.endpoints)) return undefined;
|
|
101
|
+
const endpoint = data.endpoints.api;
|
|
102
|
+
return typeof endpoint === "string" ? normalizeGitHubCopilotApiEndpoint(endpoint) : undefined;
|
|
103
|
+
} catch {
|
|
104
|
+
return undefined;
|
|
105
|
+
}
|
|
106
|
+
}
|
|
75
107
|
|
|
76
108
|
export function parseGitHubCopilotApiKey(apiKeyRaw: string): ParsedGitHubCopilotApiKey {
|
|
77
109
|
try {
|