@oh-my-pi/pi-catalog 17.3.4 → 17.3.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +29 -0
- package/dist/types/compat/openai.d.ts +2 -0
- package/dist/types/hosts.d.ts +1 -1
- package/dist/types/identity/family.d.ts +23 -0
- package/dist/types/openai-pricing.d.ts +17 -0
- package/dist/types/provider-models/descriptors.d.ts +4 -4
- package/dist/types/provider-models/openai-compat.d.ts +17 -1
- package/dist/types/types.d.ts +15 -1
- package/package.json +4 -4
- package/src/compat/openai.ts +54 -11
- package/src/discovery/codex.ts +21 -9
- package/src/hosts.ts +1 -1
- package/src/identity/family.ts +47 -1
- package/src/model-thinking.ts +20 -2
- package/src/models.json +3509 -1011
- package/src/openai-pricing.ts +29 -0
- package/src/provider-models/descriptors.ts +3 -3
- package/src/provider-models/openai-compat.ts +103 -48
- package/src/types.ts +15 -0
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
import type { TokenCost } from "./types";
|
|
2
|
+
|
|
3
|
+
/** Standard GPT-5.6 Sol rates used by the Daybreak Blue aliases. */
|
|
4
|
+
export const OPENAI_GPT_56_SOL_STANDARD_COST = {
|
|
5
|
+
input: 5,
|
|
6
|
+
output: 30,
|
|
7
|
+
cacheRead: 0.5,
|
|
8
|
+
cacheWrite: 6.25,
|
|
9
|
+
} as const satisfies TokenCost;
|
|
10
|
+
|
|
11
|
+
/** Standard GPT-5.6 Cyber rates used by the Daybreak Red aliases. */
|
|
12
|
+
export const OPENAI_GPT_56_CYBER_STANDARD_COST = {
|
|
13
|
+
input: 12.5,
|
|
14
|
+
output: 75,
|
|
15
|
+
cacheRead: 1.25,
|
|
16
|
+
cacheWrite: 15.625,
|
|
17
|
+
} as const satisfies TokenCost;
|
|
18
|
+
|
|
19
|
+
/** Resolve standard rates for Codex-prefixed Daybreak aliases. */
|
|
20
|
+
export function resolveOpenAIDaybreakStandardCost(modelId: string): TokenCost | undefined {
|
|
21
|
+
switch (modelId) {
|
|
22
|
+
case "gpt-daybreak-blue-latest":
|
|
23
|
+
return OPENAI_GPT_56_SOL_STANDARD_COST;
|
|
24
|
+
case "gpt-daybreak-red-latest":
|
|
25
|
+
return OPENAI_GPT_56_CYBER_STANDARD_COST;
|
|
26
|
+
default:
|
|
27
|
+
return undefined;
|
|
28
|
+
}
|
|
29
|
+
}
|
|
@@ -478,13 +478,13 @@ export const CATALOG_PROVIDERS = [
|
|
|
478
478
|
},
|
|
479
479
|
{
|
|
480
480
|
id: "xai",
|
|
481
|
-
defaultModel: "grok-4
|
|
481
|
+
defaultModel: "grok-4.6",
|
|
482
482
|
envVars: ["XAI_API_KEY"],
|
|
483
483
|
createModelManagerOptions: (config: ModelManagerConfig) => xaiModelManagerOptions(config),
|
|
484
484
|
},
|
|
485
485
|
{
|
|
486
486
|
id: "xai-oauth",
|
|
487
|
-
defaultModel: "grok-4.
|
|
487
|
+
defaultModel: "grok-4.6",
|
|
488
488
|
envVars: ["XAI_OAUTH_TOKEN", "XAI_API_KEY"],
|
|
489
489
|
createModelManagerOptions: (config: ModelManagerConfig) => xaiOAuthModelManagerOptions(config),
|
|
490
490
|
catalogDiscovery: {
|
|
@@ -522,7 +522,7 @@ export const CATALOG_PROVIDERS = [
|
|
|
522
522
|
},
|
|
523
523
|
{
|
|
524
524
|
id: "zai",
|
|
525
|
-
defaultModel: "glm-5.
|
|
525
|
+
defaultModel: "glm-5.3",
|
|
526
526
|
envVars: ["ZAI_API_KEY"],
|
|
527
527
|
createModelManagerOptions: (config: ModelManagerConfig) => zaiModelManagerOptions(config),
|
|
528
528
|
catalogDiscovery: { label: "zAI" },
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { USER_AGENT } from "@oh-my-pi/pi-utils";
|
|
2
2
|
import * as logger from "@oh-my-pi/pi-utils/logger";
|
|
3
|
+
import { xaiResponsesReasoningEffortMap } from "../compat/openai";
|
|
3
4
|
import {
|
|
4
5
|
DEFAULT_OPENAI_COMPATIBLE_DISCOVERY_TIMEOUT_MS,
|
|
5
6
|
fetchOpenAICompatibleModels,
|
|
@@ -20,7 +21,17 @@ import {
|
|
|
20
21
|
import { resolveModelReference } from "../identity/reference";
|
|
21
22
|
import type { ModelManagerOptions } from "../model-manager";
|
|
22
23
|
import { type GeneratedProvider, getBundledModels } from "../models";
|
|
23
|
-
import
|
|
24
|
+
import { OPENAI_GPT_56_CYBER_STANDARD_COST, OPENAI_GPT_56_SOL_STANDARD_COST } from "../openai-pricing";
|
|
25
|
+
import type {
|
|
26
|
+
Api,
|
|
27
|
+
FetchImpl,
|
|
28
|
+
LongContextTokenCost,
|
|
29
|
+
Model,
|
|
30
|
+
ModelSpec,
|
|
31
|
+
OpenAICompat,
|
|
32
|
+
Provider,
|
|
33
|
+
ThinkingConfig,
|
|
34
|
+
} from "../types";
|
|
24
35
|
import { discoveryFetch, isAnthropicOAuthToken, isRecord, toBoolean, toNumber, toPositiveNumber } from "../utils";
|
|
25
36
|
import { ALIBABA_TOKEN_PLAN_BASE_URL, parseAlibabaTokenPlanCredential } from "../wire/alibaba-token-plan";
|
|
26
37
|
import { coreWeaveProjectHeaders } from "../wire/coreweave";
|
|
@@ -869,6 +880,7 @@ export function umansModelManagerOptions(config?: UmansModelManagerConfig): Mode
|
|
|
869
880
|
// ---------------------------------------------------------------------------
|
|
870
881
|
|
|
871
882
|
const OPENAI_API_BASE_URL = "https://api.openai.com/v1";
|
|
883
|
+
/** GPT-5.6 rates applied when a first-party request exceeds 272K input tokens. */
|
|
872
884
|
export const OPENAI_GPT_56_LONG_CONTEXT_COSTS = {
|
|
873
885
|
luna: {
|
|
874
886
|
inputThreshold: 272_000,
|
|
@@ -891,20 +903,7 @@ export const OPENAI_GPT_56_LONG_CONTEXT_COSTS = {
|
|
|
891
903
|
cacheRead: 0.4,
|
|
892
904
|
cacheWrite: 5,
|
|
893
905
|
},
|
|
894
|
-
} as const
|
|
895
|
-
const OPENAI_GPT_56_SOL_STANDARD_COST = {
|
|
896
|
-
input: 5,
|
|
897
|
-
output: 30,
|
|
898
|
-
cacheRead: 0.5,
|
|
899
|
-
cacheWrite: 6.25,
|
|
900
|
-
longContext: OPENAI_GPT_56_LONG_CONTEXT_COSTS.sol,
|
|
901
|
-
} as const;
|
|
902
|
-
const OPENAI_GPT_56_CYBER_STANDARD_COST = {
|
|
903
|
-
input: 12.5,
|
|
904
|
-
output: 75,
|
|
905
|
-
cacheRead: 1.25,
|
|
906
|
-
cacheWrite: 15.625,
|
|
907
|
-
} as const;
|
|
906
|
+
} as const satisfies Readonly<Record<"luna" | "sol" | "terra", LongContextTokenCost>>;
|
|
908
907
|
|
|
909
908
|
export interface OpenAIModelManagerConfig {
|
|
910
909
|
apiKey?: string;
|
|
@@ -938,7 +937,10 @@ export const OPENAI_DAYBREAK_CURATED_FALLBACK_MODELS: readonly ModelSpec<"openai
|
|
|
938
937
|
baseUrl: OPENAI_API_BASE_URL,
|
|
939
938
|
reasoning: true,
|
|
940
939
|
input: ["text", "image"],
|
|
941
|
-
cost:
|
|
940
|
+
cost: {
|
|
941
|
+
...OPENAI_GPT_56_SOL_STANDARD_COST,
|
|
942
|
+
longContext: OPENAI_GPT_56_LONG_CONTEXT_COSTS.sol,
|
|
943
|
+
},
|
|
942
944
|
contextWindow: 1_050_000,
|
|
943
945
|
maxTokens: 128_000,
|
|
944
946
|
},
|
|
@@ -1257,8 +1259,23 @@ export interface XaiModelManagerConfig {
|
|
|
1257
1259
|
fetch?: FetchImpl;
|
|
1258
1260
|
}
|
|
1259
1261
|
|
|
1260
|
-
export function xaiModelManagerOptions(config?: XaiModelManagerConfig): ModelManagerOptions<"openai-
|
|
1261
|
-
return
|
|
1262
|
+
export function xaiModelManagerOptions(config?: XaiModelManagerConfig): ModelManagerOptions<"openai-responses"> {
|
|
1263
|
+
return {
|
|
1264
|
+
...createOpenAICompatibleModelManagerOptions({
|
|
1265
|
+
api: "openai-responses",
|
|
1266
|
+
providerId: "xai",
|
|
1267
|
+
defaultBaseUrl: "https://api.x.ai/v1",
|
|
1268
|
+
config,
|
|
1269
|
+
requireApiKey: true,
|
|
1270
|
+
mapModel: mapWithBundledReference,
|
|
1271
|
+
}),
|
|
1272
|
+
// Completions → Responses migration: a fresh authoritative cache written
|
|
1273
|
+
// by the old resolver stores `api: "openai-completions"` for these ids.
|
|
1274
|
+
// Without a drop list, `online-if-uncached` skips the network and
|
|
1275
|
+
// `mergeDynamicModel` lets the cached api win over the new static
|
|
1276
|
+
// Responses entries until TTL expiry.
|
|
1277
|
+
dropCachedModelIdsOnStaticMismatch: getBundledModels("xai").map(model => model.id),
|
|
1278
|
+
};
|
|
1262
1279
|
}
|
|
1263
1280
|
|
|
1264
1281
|
export interface XaiOAuthModelManagerConfig {
|
|
@@ -1316,6 +1333,7 @@ export const XAI_OAUTH_CURATED_MODELS: readonly XAICuratedModel[] = [
|
|
|
1316
1333
|
},
|
|
1317
1334
|
{ id: "grok-4.3", contextWindow: 1_000_000, name: "Grok 4.3", input: ["text", "image"] },
|
|
1318
1335
|
{ id: "grok-4.5", contextWindow: 500_000, name: "Grok 4.5", input: ["text", "image"] },
|
|
1336
|
+
{ id: "grok-4.6", contextWindow: 500_000, name: "Grok 4.6", input: ["text", "image"] },
|
|
1319
1337
|
// grok-4.20-multi-agent-0309 is text-only per the bundled catalog; omit `input` for the default.
|
|
1320
1338
|
{ id: "grok-4.20-multi-agent-0309", contextWindow: 2_000_000, name: "Grok 4.20 (Multi-Agent)" },
|
|
1321
1339
|
{
|
|
@@ -1352,21 +1370,47 @@ const XAI_NON_CHAT_PREFIXES = ["grok-imagine-", "grok-stt-", "grok-voice-"] as c
|
|
|
1352
1370
|
function withXaiOAuthCompatDefaults(model: ModelSpec<"openai-responses">): ModelSpec<"openai-responses"> {
|
|
1353
1371
|
const compat = {
|
|
1354
1372
|
...(model.compat ?? {}),
|
|
1355
|
-
includeEncryptedReasoning: model.compat?.includeEncryptedReasoning ??
|
|
1356
|
-
filterReasoningHistory: model.compat?.filterReasoningHistory ??
|
|
1373
|
+
includeEncryptedReasoning: model.compat?.includeEncryptedReasoning ?? true,
|
|
1374
|
+
filterReasoningHistory: model.compat?.filterReasoningHistory ?? false,
|
|
1357
1375
|
supportsImageDetailOriginal: model.compat?.supportsImageDetailOriginal ?? false,
|
|
1358
1376
|
omitReasoningEffort: model.compat?.omitReasoningEffort ?? !isGrokReasoningEffortCapable(model.id),
|
|
1359
1377
|
};
|
|
1360
1378
|
return { ...model, compat };
|
|
1361
1379
|
}
|
|
1362
1380
|
|
|
1363
|
-
// Hermes-agent parity
|
|
1364
|
-
//
|
|
1365
|
-
//
|
|
1366
|
-
//
|
|
1367
|
-
|
|
1368
|
-
|
|
1369
|
-
|
|
1381
|
+
// Hermes-agent parity for `minimal -> low` (see hermes-agent/agent/transports/
|
|
1382
|
+
// codex.py:92). Multi-agent Grok keeps `xhigh` unmapped (agent-count mode);
|
|
1383
|
+
// other first-party SKUs clamp leftover `xhigh`/`max` to `high`.
|
|
1384
|
+
// `resolveModelThinking` folds this into `model.thinking.effortMap`.
|
|
1385
|
+
|
|
1386
|
+
/**
|
|
1387
|
+
* Bake first-party xAI Responses effort-dial metadata onto a catalog spec.
|
|
1388
|
+
*
|
|
1389
|
+
* models.dev marks many Grok SKUs as reasoners and the thinking rebake would
|
|
1390
|
+
* otherwise emit a default `minimal/low/medium/high` dial. api.x.ai only
|
|
1391
|
+
* accepts `reasoning.effort` for {@link isGrokReasoningEffortCapable} ids —
|
|
1392
|
+
* off-allowlist reasoners (`grok-code-fast-1`, `grok-build-0.1`,
|
|
1393
|
+
* `grok-4.20-0309-reasoning`, …) 400 if the param is sent. SuperGrok
|
|
1394
|
+
* (`xai-oauth`) already curates this via {@link mergeCuratedIntoModel}; paid
|
|
1395
|
+
* `xai` rows come from stencil.so and need the same wire facts in the exported
|
|
1396
|
+
* `models.json` so direct catalog readers do not present an unsupported dial.
|
|
1397
|
+
*
|
|
1398
|
+
* Explicit `compat.supportsReasoningEffort` / `omitReasoningEffort` win.
|
|
1399
|
+
*/
|
|
1400
|
+
export function applyXaiResponsesThinkingPolicy(model: ModelSpec<"openai-responses">): ModelSpec<"openai-responses"> {
|
|
1401
|
+
const effortCapable = model.compat?.supportsReasoningEffort ?? isGrokReasoningEffortCapable(model.id);
|
|
1402
|
+
const compat = {
|
|
1403
|
+
...(model.compat ?? {}),
|
|
1404
|
+
supportsReasoningEffort: effortCapable,
|
|
1405
|
+
omitReasoningEffort: model.compat?.omitReasoningEffort ?? !effortCapable,
|
|
1406
|
+
};
|
|
1407
|
+
if (effortCapable) {
|
|
1408
|
+
compat.reasoningEffortMap = { ...xaiResponsesReasoningEffortMap(model.id) };
|
|
1409
|
+
} else {
|
|
1410
|
+
delete compat.reasoningEffortMap;
|
|
1411
|
+
}
|
|
1412
|
+
return { ...model, compat };
|
|
1413
|
+
}
|
|
1370
1414
|
|
|
1371
1415
|
// xai-oauth's /v1/models exposes no per-request output limit on the OAuth
|
|
1372
1416
|
// (Grok Build / SuperGrok) surface, so the curated catalog owns `maxTokens`
|
|
@@ -1381,9 +1425,9 @@ const XAI_REASONING_EFFORT_MAP = { minimal: "low" } as const;
|
|
|
1381
1425
|
// reasoning metadata and fetchOpenAICompatibleModels defaults reasoning to
|
|
1382
1426
|
// false). Caller supplies a `base` Model (either a freshly synthesised seed
|
|
1383
1427
|
// or a dynamic-fetched entry); the helper layers curated fields on top.
|
|
1384
|
-
// The
|
|
1385
|
-
//
|
|
1386
|
-
//
|
|
1428
|
+
// The effort remap from {@link xaiResponsesReasoningEffortMap} is merged
|
|
1429
|
+
// only onto effort-capable rows. Off-allowlist reasoners omit the wire
|
|
1430
|
+
// param, so a map on those specs is dead weight.
|
|
1387
1431
|
// The effort-dial pair (`supportsReasoningEffort`/`omitReasoningEffort`) is
|
|
1388
1432
|
// authoritative: a stale flag on `base` (previous snapshot or dynamic fetch)
|
|
1389
1433
|
// must not outlive an allowlist change in identity/family.ts.
|
|
@@ -1394,13 +1438,17 @@ function mergeCuratedIntoModel(
|
|
|
1394
1438
|
const effortCapable = curated.supportsReasoningEffort ?? isGrokReasoningEffortCapable(curated.id);
|
|
1395
1439
|
const compat = {
|
|
1396
1440
|
...(base.compat ?? {}),
|
|
1397
|
-
|
|
1398
|
-
|
|
1399
|
-
filterReasoningHistory: base.compat?.filterReasoningHistory ?? true,
|
|
1441
|
+
includeEncryptedReasoning: base.compat?.includeEncryptedReasoning ?? true,
|
|
1442
|
+
filterReasoningHistory: false,
|
|
1400
1443
|
supportsImageDetailOriginal: base.compat?.supportsImageDetailOriginal ?? false,
|
|
1401
1444
|
omitReasoningEffort: !effortCapable,
|
|
1402
1445
|
supportsReasoningEffort: effortCapable,
|
|
1403
1446
|
};
|
|
1447
|
+
if (effortCapable) {
|
|
1448
|
+
compat.reasoningEffortMap = { ...xaiResponsesReasoningEffortMap(curated.id) };
|
|
1449
|
+
} else {
|
|
1450
|
+
delete compat.reasoningEffortMap;
|
|
1451
|
+
}
|
|
1404
1452
|
return {
|
|
1405
1453
|
...base,
|
|
1406
1454
|
contextWindow: curated.contextWindow,
|
|
@@ -1502,7 +1550,7 @@ export function buildXaiOAuthStaticSeed(baseUrl?: string): ModelSpec<"openai-res
|
|
|
1502
1550
|
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
|
1503
1551
|
contextWindow: curated.contextWindow,
|
|
1504
1552
|
maxTokens: curated.contextWindow,
|
|
1505
|
-
compat: { reasoningEffortMap:
|
|
1553
|
+
compat: { reasoningEffortMap: xaiResponsesReasoningEffortMap(curated.id) },
|
|
1506
1554
|
};
|
|
1507
1555
|
return mergeCuratedIntoModel(base, curated);
|
|
1508
1556
|
});
|
|
@@ -3634,15 +3682,19 @@ export function basetenModelManagerOptions(
|
|
|
3634
3682
|
const features = Array.isArray(raw.supported_features) ? raw.supported_features : [];
|
|
3635
3683
|
const modalities = Array.isArray(raw.input_modalities) ? raw.input_modalities : [];
|
|
3636
3684
|
|
|
3637
|
-
// Baseten's
|
|
3638
|
-
//
|
|
3639
|
-
|
|
3685
|
+
// Baseten's discovery flags are not enough to enable OMP reasoning for every
|
|
3686
|
+
// model. Only models with a verified Baseten reasoning policy are enabled
|
|
3687
|
+
// here; an unknown model may use a different reasoning wire shape or effort
|
|
3688
|
+
// vocabulary, which OMP must not guess.
|
|
3689
|
+
const isSupportedBasetenReasoningModel =
|
|
3690
|
+
isKimiK3ModelId(defaults.id) ||
|
|
3640
3691
|
defaults.id === "openai/gpt-oss-120b" ||
|
|
3692
|
+
defaults.id === "deepseek-ai/DeepSeek-V4-Pro" ||
|
|
3641
3693
|
defaults.id === "zai-org/GLM-5.2" ||
|
|
3642
3694
|
defaults.id === "zai-org/GLM-5.2-Fast";
|
|
3643
|
-
const isBasetenNativeReasoning = isEffortReasoning || defaults.id === "deepseek-ai/DeepSeek-V4-Pro";
|
|
3644
3695
|
const reasoning =
|
|
3645
|
-
|
|
3696
|
+
isSupportedBasetenReasoningModel &&
|
|
3697
|
+
(features.includes("reasoning") || features.includes("reasoning_effort"));
|
|
3646
3698
|
const supportsTools = features.includes("tools") ? undefined : false;
|
|
3647
3699
|
const vision = modalities.includes("image") || (reference?.input.includes("image") ?? false);
|
|
3648
3700
|
|
|
@@ -3656,14 +3708,7 @@ export function basetenModelManagerOptions(
|
|
|
3656
3708
|
|
|
3657
3709
|
const contextWindow = toPositiveNumber(raw.context_length, reference?.contextWindow ?? defaults.contextWindow);
|
|
3658
3710
|
const maxTokens = toPositiveNumber(raw.max_completion_tokens, reference?.maxTokens ?? defaults.maxTokens);
|
|
3659
|
-
|
|
3660
3711
|
const baseModel = mapWithBundledReference(entry, defaults, reference);
|
|
3661
|
-
const thinking = isEffortReasoning
|
|
3662
|
-
? {
|
|
3663
|
-
mode: "effort" as const,
|
|
3664
|
-
efforts: [Effort.High, Effort.Max],
|
|
3665
|
-
}
|
|
3666
|
-
: undefined;
|
|
3667
3712
|
|
|
3668
3713
|
return {
|
|
3669
3714
|
...baseModel,
|
|
@@ -3672,7 +3717,6 @@ export function basetenModelManagerOptions(
|
|
|
3672
3717
|
cost,
|
|
3673
3718
|
contextWindow,
|
|
3674
3719
|
maxTokens,
|
|
3675
|
-
...(thinking ? { thinking } : {}),
|
|
3676
3720
|
...(supportsTools === false ? { supportsTools } : {}),
|
|
3677
3721
|
};
|
|
3678
3722
|
},
|
|
@@ -5713,6 +5757,15 @@ function openAiCompletionsDescriptor(
|
|
|
5713
5757
|
return simpleModelsDevDescriptor(modelsDevKey, providerId, "openai-completions", baseUrl, options);
|
|
5714
5758
|
}
|
|
5715
5759
|
|
|
5760
|
+
function openAiResponsesDescriptor(
|
|
5761
|
+
modelsDevKey: string,
|
|
5762
|
+
providerId: string,
|
|
5763
|
+
baseUrl: string,
|
|
5764
|
+
options: Omit<ModelsDevProviderDescriptor, "modelsDevKey" | "providerId" | "api" | "baseUrl"> = {},
|
|
5765
|
+
): ModelsDevProviderDescriptor {
|
|
5766
|
+
return simpleModelsDevDescriptor(modelsDevKey, providerId, "openai-responses", baseUrl, options);
|
|
5767
|
+
}
|
|
5768
|
+
|
|
5716
5769
|
function anthropicMessagesDescriptor(
|
|
5717
5770
|
modelsDevKey: string,
|
|
5718
5771
|
providerId: string,
|
|
@@ -5837,7 +5890,9 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
|
|
|
5837
5890
|
defaultContextWindow: 131072,
|
|
5838
5891
|
}),
|
|
5839
5892
|
// --- xAI ---
|
|
5840
|
-
|
|
5893
|
+
openAiResponsesDescriptor("xai", "xai", "https://api.x.ai/v1", {
|
|
5894
|
+
transformModel: model => applyXaiResponsesThinkingPolicy(model as ModelSpec<"openai-responses">),
|
|
5895
|
+
}),
|
|
5841
5896
|
// --- DeepSeek ---
|
|
5842
5897
|
openAiCompletionsDescriptor("deepseek", "deepseek", "https://api.deepseek.com", {
|
|
5843
5898
|
// Only ship the v4 family as built-ins; older deepseek-chat / deepseek-reasoner
|
package/src/types.ts
CHANGED
|
@@ -363,6 +363,13 @@ export interface OpenAICompat {
|
|
|
363
363
|
* model id. Default: true. Issue #5606.
|
|
364
364
|
*/
|
|
365
365
|
supportsSamplingParams?: boolean;
|
|
366
|
+
/**
|
|
367
|
+
* Whether presence/frequency penalties and stop sequences may be sent.
|
|
368
|
+
* First-party xAI `/v1/responses` rejects penalty fields for every model.
|
|
369
|
+
* xAI reasoning models also reject them (and `stop`) on chat completions.
|
|
370
|
+
* When unset, auto-detected. Default: true.
|
|
371
|
+
*/
|
|
372
|
+
supportsPenaltyAndStopParams?: boolean;
|
|
366
373
|
/** Always send a max-token field when the caller did not provide one. Default: auto-detected (Kimi-family models derive TPM limits from max_tokens). */
|
|
367
374
|
alwaysSendMaxTokens?: boolean;
|
|
368
375
|
/** Whether Responses-API tool-call/result history must be strictly paired. Default: auto-detected (Azure OpenAI, GitHub Copilot). */
|
|
@@ -578,6 +585,7 @@ export interface ResolvedOpenAISharedCompat {
|
|
|
578
585
|
reasoningEffortMap: Partial<Record<Effort, string>>;
|
|
579
586
|
supportsReasoningParams: boolean;
|
|
580
587
|
supportsSamplingParams: boolean;
|
|
588
|
+
supportsPenaltyAndStopParams: boolean;
|
|
581
589
|
thinkingFormat: OpenAIReasoningFormat;
|
|
582
590
|
/** Kimi Code transport selected by live per-model protocol metadata. */
|
|
583
591
|
kimiApiFormat?: OpenAICompat["kimiApiFormat"];
|
|
@@ -643,6 +651,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
|
|
|
643
651
|
| "reasoningEffortMap"
|
|
644
652
|
| "supportsReasoningParams"
|
|
645
653
|
| "supportsSamplingParams"
|
|
654
|
+
| "supportsPenaltyAndStopParams"
|
|
646
655
|
| "thinkingFormat"
|
|
647
656
|
| "kimiApiFormat"
|
|
648
657
|
| "reasoningDisableMode"
|
|
@@ -711,6 +720,12 @@ export interface ResolvedOpenAIResponsesCompat extends ResolvedOpenAISharedCompa
|
|
|
711
720
|
strictResponsesPairing: boolean;
|
|
712
721
|
supportsImageDetailOriginal: boolean;
|
|
713
722
|
supportsObfuscationOptOut: boolean;
|
|
723
|
+
/**
|
|
724
|
+
* Whether `reasoning.summary` may be sent. First-party xAI `/v1/responses`
|
|
725
|
+
* rejects the field; handlers pass `null` so the wire omits it instead of
|
|
726
|
+
* filling `"auto"`.
|
|
727
|
+
*/
|
|
728
|
+
supportsReasoningSummary: boolean;
|
|
714
729
|
streamIdleTimeoutMs?: number;
|
|
715
730
|
vercelGatewayRouting?: OpenAICompat["vercelGatewayRouting"];
|
|
716
731
|
/** The model sits behind Vercel AI Gateway's Responses endpoint. */
|