@oh-my-pi/pi-catalog 17.3.3 → 17.3.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,29 @@
1
+ import type { TokenCost } from "./types";
2
+
3
+ /** Standard GPT-5.6 Sol rates used by the Daybreak Blue aliases. */
4
+ export const OPENAI_GPT_56_SOL_STANDARD_COST = {
5
+ input: 5,
6
+ output: 30,
7
+ cacheRead: 0.5,
8
+ cacheWrite: 6.25,
9
+ } as const satisfies TokenCost;
10
+
11
+ /** Standard GPT-5.6 Cyber rates used by the Daybreak Red aliases. */
12
+ export const OPENAI_GPT_56_CYBER_STANDARD_COST = {
13
+ input: 12.5,
14
+ output: 75,
15
+ cacheRead: 1.25,
16
+ cacheWrite: 15.625,
17
+ } as const satisfies TokenCost;
18
+
19
+ /** Resolve standard rates for Codex-prefixed Daybreak aliases. */
20
+ export function resolveOpenAIDaybreakStandardCost(modelId: string): TokenCost | undefined {
21
+ switch (modelId) {
22
+ case "gpt-daybreak-blue-latest":
23
+ return OPENAI_GPT_56_SOL_STANDARD_COST;
24
+ case "gpt-daybreak-red-latest":
25
+ return OPENAI_GPT_56_CYBER_STANDARD_COST;
26
+ default:
27
+ return undefined;
28
+ }
29
+ }
@@ -1,3 +1,5 @@
1
+ import { PERSONAL_GITHUB_COPILOT_BASE_URL } from "../wire/github-copilot";
2
+
1
3
  export interface ModelCacheProviderIdOptions {
2
4
  apiKey?: string;
3
5
  baseUrl?: string;
@@ -56,6 +58,18 @@ export function resolveModelCacheProviderId(providerId: string, options: ModelCa
56
58
  const scope = `${options.apiKey ?? ""}\u0000${discoveryBaseUrl}`;
57
59
  return `${providerId}:models-v1:${Bun.hash(scope).toString(36)}`;
58
60
  }
61
+ case "github-copilot": {
62
+ // Copilot model specs bake in the plan-specific endpoint (personal vs
63
+ // Business/Enterprise) resolved from the credential. Discovery writes an
64
+ // authoritative cache, so `online-if-uncached` serves it for the full
65
+ // TTL without re-probing. Keying the namespace on the credential means
66
+ // switching `COPILOT_GITHUB_TOKEN` to a different account misses the
67
+ // prior endpoint's cache and re-runs discovery instead of hitting the
68
+ // stale host and 403ing (PR #8510 review).
69
+ const baseUrl = options.baseUrl ?? PERSONAL_GITHUB_COPILOT_BASE_URL;
70
+ const scope = `${options.apiKey ?? ""}\u0000${baseUrl}`;
71
+ return `github-copilot:models-v1:${Bun.hash(scope).toString(36)}`;
72
+ }
59
73
  case "openrouter":
60
74
  return "openrouter:pseudo-api";
61
75
  case "vllm": {
@@ -478,13 +478,13 @@ export const CATALOG_PROVIDERS = [
478
478
  },
479
479
  {
480
480
  id: "xai",
481
- defaultModel: "grok-4-fast-non-reasoning",
481
+ defaultModel: "grok-4.5",
482
482
  envVars: ["XAI_API_KEY"],
483
483
  createModelManagerOptions: (config: ModelManagerConfig) => xaiModelManagerOptions(config),
484
484
  },
485
485
  {
486
486
  id: "xai-oauth",
487
- defaultModel: "grok-4.3",
487
+ defaultModel: "grok-4.5",
488
488
  envVars: ["XAI_OAUTH_TOKEN", "XAI_API_KEY"],
489
489
  createModelManagerOptions: (config: ModelManagerConfig) => xaiOAuthModelManagerOptions(config),
490
490
  catalogDiscovery: {
@@ -522,7 +522,7 @@ export const CATALOG_PROVIDERS = [
522
522
  },
523
523
  {
524
524
  id: "zai",
525
- defaultModel: "glm-5.2",
525
+ defaultModel: "glm-5.3",
526
526
  envVars: ["ZAI_API_KEY"],
527
527
  createModelManagerOptions: (config: ModelManagerConfig) => zaiModelManagerOptions(config),
528
528
  catalogDiscovery: { label: "zAI" },
@@ -1,6 +1,8 @@
1
1
  import { USER_AGENT } from "@oh-my-pi/pi-utils";
2
2
  import * as logger from "@oh-my-pi/pi-utils/logger";
3
+ import { xaiResponsesReasoningEffortMap } from "../compat/openai";
3
4
  import {
5
+ DEFAULT_OPENAI_COMPATIBLE_DISCOVERY_TIMEOUT_MS,
4
6
  fetchOpenAICompatibleModels,
5
7
  type OpenAICompatibleModelMapperContext,
6
8
  type OpenAICompatibleModelRecord,
@@ -19,12 +21,23 @@ import {
19
21
  import { resolveModelReference } from "../identity/reference";
20
22
  import type { ModelManagerOptions } from "../model-manager";
21
23
  import { type GeneratedProvider, getBundledModels } from "../models";
22
- import type { Api, FetchImpl, Model, ModelSpec, OpenAICompat, Provider, ThinkingConfig } from "../types";
24
+ import { OPENAI_GPT_56_CYBER_STANDARD_COST, OPENAI_GPT_56_SOL_STANDARD_COST } from "../openai-pricing";
25
+ import type {
26
+ Api,
27
+ FetchImpl,
28
+ LongContextTokenCost,
29
+ Model,
30
+ ModelSpec,
31
+ OpenAICompat,
32
+ Provider,
33
+ ThinkingConfig,
34
+ } from "../types";
23
35
  import { discoveryFetch, isAnthropicOAuthToken, isRecord, toBoolean, toNumber, toPositiveNumber } from "../utils";
24
36
  import { ALIBABA_TOKEN_PLAN_BASE_URL, parseAlibabaTokenPlanCredential } from "../wire/alibaba-token-plan";
25
37
  import { coreWeaveProjectHeaders } from "../wire/coreweave";
26
38
  import {
27
39
  COPILOT_API_HEADERS,
40
+ discoverGitHubCopilotApiEndpoint,
28
41
  getGitHubCopilotBaseUrl,
29
42
  isPersonalGitHubCopilotBaseUrl,
30
43
  parseGitHubCopilotApiKey,
@@ -867,6 +880,7 @@ export function umansModelManagerOptions(config?: UmansModelManagerConfig): Mode
867
880
  // ---------------------------------------------------------------------------
868
881
 
869
882
  const OPENAI_API_BASE_URL = "https://api.openai.com/v1";
883
+ /** GPT-5.6 rates applied when a first-party request exceeds 272K input tokens. */
870
884
  export const OPENAI_GPT_56_LONG_CONTEXT_COSTS = {
871
885
  luna: {
872
886
  inputThreshold: 272_000,
@@ -889,20 +903,7 @@ export const OPENAI_GPT_56_LONG_CONTEXT_COSTS = {
889
903
  cacheRead: 0.4,
890
904
  cacheWrite: 5,
891
905
  },
892
- } as const;
893
- const OPENAI_GPT_56_SOL_STANDARD_COST = {
894
- input: 5,
895
- output: 30,
896
- cacheRead: 0.5,
897
- cacheWrite: 6.25,
898
- longContext: OPENAI_GPT_56_LONG_CONTEXT_COSTS.sol,
899
- } as const;
900
- const OPENAI_GPT_56_CYBER_STANDARD_COST = {
901
- input: 12.5,
902
- output: 75,
903
- cacheRead: 1.25,
904
- cacheWrite: 15.625,
905
- } as const;
906
+ } as const satisfies Readonly<Record<"luna" | "sol" | "terra", LongContextTokenCost>>;
906
907
 
907
908
  export interface OpenAIModelManagerConfig {
908
909
  apiKey?: string;
@@ -936,7 +937,10 @@ export const OPENAI_DAYBREAK_CURATED_FALLBACK_MODELS: readonly ModelSpec<"openai
936
937
  baseUrl: OPENAI_API_BASE_URL,
937
938
  reasoning: true,
938
939
  input: ["text", "image"],
939
- cost: OPENAI_GPT_56_SOL_STANDARD_COST,
940
+ cost: {
941
+ ...OPENAI_GPT_56_SOL_STANDARD_COST,
942
+ longContext: OPENAI_GPT_56_LONG_CONTEXT_COSTS.sol,
943
+ },
940
944
  contextWindow: 1_050_000,
941
945
  maxTokens: 128_000,
942
946
  },
@@ -1255,8 +1259,23 @@ export interface XaiModelManagerConfig {
1255
1259
  fetch?: FetchImpl;
1256
1260
  }
1257
1261
 
1258
- export function xaiModelManagerOptions(config?: XaiModelManagerConfig): ModelManagerOptions<"openai-completions"> {
1259
- return createSimpleOpenAICompletionsOptions("xai", "https://api.x.ai/v1", config);
1262
+ export function xaiModelManagerOptions(config?: XaiModelManagerConfig): ModelManagerOptions<"openai-responses"> {
1263
+ return {
1264
+ ...createOpenAICompatibleModelManagerOptions({
1265
+ api: "openai-responses",
1266
+ providerId: "xai",
1267
+ defaultBaseUrl: "https://api.x.ai/v1",
1268
+ config,
1269
+ requireApiKey: true,
1270
+ mapModel: mapWithBundledReference,
1271
+ }),
1272
+ // Completions → Responses migration: a fresh authoritative cache written
1273
+ // by the old resolver stores `api: "openai-completions"` for these ids.
1274
+ // Without a drop list, `online-if-uncached` skips the network and
1275
+ // `mergeDynamicModel` lets the cached api win over the new static
1276
+ // Responses entries until TTL expiry.
1277
+ dropCachedModelIdsOnStaticMismatch: getBundledModels("xai").map(model => model.id),
1278
+ };
1260
1279
  }
1261
1280
 
1262
1281
  export interface XaiOAuthModelManagerConfig {
@@ -1314,6 +1333,7 @@ export const XAI_OAUTH_CURATED_MODELS: readonly XAICuratedModel[] = [
1314
1333
  },
1315
1334
  { id: "grok-4.3", contextWindow: 1_000_000, name: "Grok 4.3", input: ["text", "image"] },
1316
1335
  { id: "grok-4.5", contextWindow: 500_000, name: "Grok 4.5", input: ["text", "image"] },
1336
+ { id: "grok-4.6", contextWindow: 500_000, name: "Grok 4.6", input: ["text", "image"] },
1317
1337
  // grok-4.20-multi-agent-0309 is text-only per the bundled catalog; omit `input` for the default.
1318
1338
  { id: "grok-4.20-multi-agent-0309", contextWindow: 2_000_000, name: "Grok 4.20 (Multi-Agent)" },
1319
1339
  {
@@ -1350,21 +1370,47 @@ const XAI_NON_CHAT_PREFIXES = ["grok-imagine-", "grok-stt-", "grok-voice-"] as c
1350
1370
  function withXaiOAuthCompatDefaults(model: ModelSpec<"openai-responses">): ModelSpec<"openai-responses"> {
1351
1371
  const compat = {
1352
1372
  ...(model.compat ?? {}),
1353
- includeEncryptedReasoning: model.compat?.includeEncryptedReasoning ?? false,
1354
- filterReasoningHistory: model.compat?.filterReasoningHistory ?? true,
1373
+ includeEncryptedReasoning: model.compat?.includeEncryptedReasoning ?? true,
1374
+ filterReasoningHistory: model.compat?.filterReasoningHistory ?? false,
1355
1375
  supportsImageDetailOriginal: model.compat?.supportsImageDetailOriginal ?? false,
1356
1376
  omitReasoningEffort: model.compat?.omitReasoningEffort ?? !isGrokReasoningEffortCapable(model.id),
1357
1377
  };
1358
1378
  return { ...model, compat };
1359
1379
  }
1360
1380
 
1361
- // Hermes-agent parity: only the `minimal -> low` clamp is applied (see
1362
- // hermes-agent/agent/transports/codex.py:92 `_effort_clamp = {"minimal":
1363
- // "low"}`). Hermes sends `xhigh` to xAI verbatim and we match that contract
1364
- // — let xAI decide if the level is valid for the specific Grok model.
1365
- // `resolveModelThinking` folds this into `model.thinking.effortMap`, downstream
1366
- // of the omitReasoningEffort gate in pi-ai's stream.ts.
1367
- const XAI_REASONING_EFFORT_MAP = { minimal: "low" } as const;
1381
+ // Hermes-agent parity for `minimal -> low` (see hermes-agent/agent/transports/
1382
+ // codex.py:92). Multi-agent Grok keeps `xhigh` unmapped (agent-count mode);
1383
+ // other first-party SKUs clamp leftover `xhigh`/`max` to `high`.
1384
+ // `resolveModelThinking` folds this into `model.thinking.effortMap`.
1385
+
1386
+ /**
1387
+ * Bake first-party xAI Responses effort-dial metadata onto a catalog spec.
1388
+ *
1389
+ * models.dev marks many Grok SKUs as reasoners and the thinking rebake would
1390
+ * otherwise emit a default `minimal/low/medium/high` dial. api.x.ai only
1391
+ * accepts `reasoning.effort` for {@link isGrokReasoningEffortCapable} ids —
1392
+ * off-allowlist reasoners (`grok-code-fast-1`, `grok-build-0.1`,
1393
+ * `grok-4.20-0309-reasoning`, …) 400 if the param is sent. SuperGrok
1394
+ * (`xai-oauth`) already curates this via {@link mergeCuratedIntoModel}; paid
1395
+ * `xai` rows come from stencil.so and need the same wire facts in the exported
1396
+ * `models.json` so direct catalog readers do not present an unsupported dial.
1397
+ *
1398
+ * Explicit `compat.supportsReasoningEffort` / `omitReasoningEffort` win.
1399
+ */
1400
+ export function applyXaiResponsesThinkingPolicy(model: ModelSpec<"openai-responses">): ModelSpec<"openai-responses"> {
1401
+ const effortCapable = model.compat?.supportsReasoningEffort ?? isGrokReasoningEffortCapable(model.id);
1402
+ const compat = {
1403
+ ...(model.compat ?? {}),
1404
+ supportsReasoningEffort: effortCapable,
1405
+ omitReasoningEffort: model.compat?.omitReasoningEffort ?? !effortCapable,
1406
+ };
1407
+ if (effortCapable) {
1408
+ compat.reasoningEffortMap = { ...xaiResponsesReasoningEffortMap(model.id) };
1409
+ } else {
1410
+ delete compat.reasoningEffortMap;
1411
+ }
1412
+ return { ...model, compat };
1413
+ }
1368
1414
 
1369
1415
  // xai-oauth's /v1/models exposes no per-request output limit on the OAuth
1370
1416
  // (Grok Build / SuperGrok) surface, so the curated catalog owns `maxTokens`
@@ -1379,9 +1425,9 @@ const XAI_REASONING_EFFORT_MAP = { minimal: "low" } as const;
1379
1425
  // reasoning metadata and fetchOpenAICompatibleModels defaults reasoning to
1380
1426
  // false). Caller supplies a `base` Model (either a freshly synthesised seed
1381
1427
  // or a dynamic-fetched entry); the helper layers curated fields on top.
1382
- // The `minimal -> low` effort clamp (XAI_REASONING_EFFORT_MAP) is always
1383
- // merged in so dynamic-fetched models — which arrive without curated
1384
- // compat keys — still get the clamp applyResponsesReasoningParams expects.
1428
+ // The effort remap from {@link xaiResponsesReasoningEffortMap} is merged
1429
+ // only onto effort-capable rows. Off-allowlist reasoners omit the wire
1430
+ // param, so a map on those specs is dead weight.
1385
1431
  // The effort-dial pair (`supportsReasoningEffort`/`omitReasoningEffort`) is
1386
1432
  // authoritative: a stale flag on `base` (previous snapshot or dynamic fetch)
1387
1433
  // must not outlive an allowlist change in identity/family.ts.
@@ -1392,13 +1438,17 @@ function mergeCuratedIntoModel(
1392
1438
  const effortCapable = curated.supportsReasoningEffort ?? isGrokReasoningEffortCapable(curated.id);
1393
1439
  const compat = {
1394
1440
  ...(base.compat ?? {}),
1395
- reasoningEffortMap: { ...XAI_REASONING_EFFORT_MAP, ...(base.compat?.reasoningEffortMap ?? {}) },
1396
- includeEncryptedReasoning: base.compat?.includeEncryptedReasoning ?? false,
1397
- filterReasoningHistory: base.compat?.filterReasoningHistory ?? true,
1441
+ includeEncryptedReasoning: base.compat?.includeEncryptedReasoning ?? true,
1442
+ filterReasoningHistory: false,
1398
1443
  supportsImageDetailOriginal: base.compat?.supportsImageDetailOriginal ?? false,
1399
1444
  omitReasoningEffort: !effortCapable,
1400
1445
  supportsReasoningEffort: effortCapable,
1401
1446
  };
1447
+ if (effortCapable) {
1448
+ compat.reasoningEffortMap = { ...xaiResponsesReasoningEffortMap(curated.id) };
1449
+ } else {
1450
+ delete compat.reasoningEffortMap;
1451
+ }
1402
1452
  return {
1403
1453
  ...base,
1404
1454
  contextWindow: curated.contextWindow,
@@ -1500,7 +1550,7 @@ export function buildXaiOAuthStaticSeed(baseUrl?: string): ModelSpec<"openai-res
1500
1550
  cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
1501
1551
  contextWindow: curated.contextWindow,
1502
1552
  maxTokens: curated.contextWindow,
1503
- compat: { reasoningEffortMap: XAI_REASONING_EFFORT_MAP },
1553
+ compat: { reasoningEffortMap: xaiResponsesReasoningEffortMap(curated.id) },
1504
1554
  };
1505
1555
  return mergeCuratedIntoModel(base, curated);
1506
1556
  });
@@ -3632,15 +3682,19 @@ export function basetenModelManagerOptions(
3632
3682
  const features = Array.isArray(raw.supported_features) ? raw.supported_features : [];
3633
3683
  const modalities = Array.isArray(raw.input_modalities) ? raw.input_modalities : [];
3634
3684
 
3635
- // Baseten's reasoning router accepts only the high/max
3636
- // effort tiers for its GLM-5.2 and gpt-oss routes.
3637
- const isEffortReasoning =
3685
+ // Baseten's discovery flags are not enough to enable OMP reasoning for every
3686
+ // model. Only models with a verified Baseten reasoning policy are enabled
3687
+ // here; an unknown model may use a different reasoning wire shape or effort
3688
+ // vocabulary, which OMP must not guess.
3689
+ const isSupportedBasetenReasoningModel =
3690
+ isKimiK3ModelId(defaults.id) ||
3638
3691
  defaults.id === "openai/gpt-oss-120b" ||
3692
+ defaults.id === "deepseek-ai/DeepSeek-V4-Pro" ||
3639
3693
  defaults.id === "zai-org/GLM-5.2" ||
3640
3694
  defaults.id === "zai-org/GLM-5.2-Fast";
3641
- const isBasetenNativeReasoning = isEffortReasoning || defaults.id === "deepseek-ai/DeepSeek-V4-Pro";
3642
3695
  const reasoning =
3643
- isBasetenNativeReasoning && (features.includes("reasoning") || features.includes("reasoning_effort"));
3696
+ isSupportedBasetenReasoningModel &&
3697
+ (features.includes("reasoning") || features.includes("reasoning_effort"));
3644
3698
  const supportsTools = features.includes("tools") ? undefined : false;
3645
3699
  const vision = modalities.includes("image") || (reference?.input.includes("image") ?? false);
3646
3700
 
@@ -3654,14 +3708,7 @@ export function basetenModelManagerOptions(
3654
3708
 
3655
3709
  const contextWindow = toPositiveNumber(raw.context_length, reference?.contextWindow ?? defaults.contextWindow);
3656
3710
  const maxTokens = toPositiveNumber(raw.max_completion_tokens, reference?.maxTokens ?? defaults.maxTokens);
3657
-
3658
3711
  const baseModel = mapWithBundledReference(entry, defaults, reference);
3659
- const thinking = isEffortReasoning
3660
- ? {
3661
- mode: "effort" as const,
3662
- efforts: [Effort.High, Effort.Max],
3663
- }
3664
- : undefined;
3665
3712
 
3666
3713
  return {
3667
3714
  ...baseModel,
@@ -3670,7 +3717,6 @@ export function basetenModelManagerOptions(
3670
3717
  cost,
3671
3718
  contextWindow,
3672
3719
  maxTokens,
3673
- ...(thinking ? { thinking } : {}),
3674
3720
  ...(supportsTools === false ? { supportsTools } : {}),
3675
3721
  };
3676
3722
  },
@@ -5191,6 +5237,7 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
5191
5237
  const resolveReference = createReferenceResolver(getProviderReferences);
5192
5238
  return {
5193
5239
  providerId: "github-copilot",
5240
+ cacheProviderId: resolveModelCacheProviderId("github-copilot", { apiKey: rawApiKey, baseUrl }),
5194
5241
  dropCachedModelIdsOnStaticMismatch: COPILOT_CACHE_INVALIDATED_MODEL_IDS,
5195
5242
  // COPILOT_API_HEADERS are compile-time constants (User-Agent + API
5196
5243
  // version), not credentials. The cache omits all request headers for
@@ -5201,11 +5248,17 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
5201
5248
  restorableHeaderFallback: { ...COPILOT_API_HEADERS },
5202
5249
  ...(apiKey && {
5203
5250
  fetchDynamicModels: async () => {
5251
+ const fetchImpl = discoveryFetch(config?.fetch);
5252
+ const requestBaseUrl = isPersonalGitHubCopilotBaseUrl(baseUrl)
5253
+ ? ((await withCatalogDiscoveryTimeout(DEFAULT_OPENAI_COMPATIBLE_DISCOVERY_TIMEOUT_MS, signal =>
5254
+ discoverGitHubCopilotApiEndpoint(apiKey, fetchImpl, signal),
5255
+ )) ?? baseUrl)
5256
+ : baseUrl;
5204
5257
  const longContextVariants: ModelSpec<Api>[] = [];
5205
5258
  const models = await fetchOpenAICompatibleModels<Api>({
5206
5259
  api: "openai-completions",
5207
5260
  provider: "github-copilot",
5208
- baseUrl,
5261
+ baseUrl: requestBaseUrl,
5209
5262
  apiKey,
5210
5263
  headers: COPILOT_API_HEADERS,
5211
5264
  mapModel: (
@@ -5251,7 +5304,7 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
5251
5304
  const input: ModelSpec<Api>["input"] =
5252
5305
  supportsVision === true
5253
5306
  ? ["text", "image"]
5254
- : supportsVision === false || !isPersonalGitHubCopilotBaseUrl(baseUrl)
5307
+ : supportsVision === false || !isPersonalGitHubCopilotBaseUrl(requestBaseUrl)
5255
5308
  ? ["text"]
5256
5309
  : (reference?.input ?? defaults.input);
5257
5310
  // With COPILOT_API_HEADERS the served window is the long-context
@@ -5272,7 +5325,7 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
5272
5325
  ...reference,
5273
5326
  api,
5274
5327
  provider: "github-copilot",
5275
- baseUrl,
5328
+ baseUrl: requestBaseUrl,
5276
5329
  name,
5277
5330
  input,
5278
5331
  contextWindow: defaultTierWindow,
@@ -5294,7 +5347,7 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
5294
5347
  : {
5295
5348
  ...defaults,
5296
5349
  api,
5297
- baseUrl,
5350
+ baseUrl: requestBaseUrl,
5298
5351
  name,
5299
5352
  input,
5300
5353
  contextWindow: defaultTierWindow,
@@ -5340,7 +5393,7 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
5340
5393
  }
5341
5394
  return base;
5342
5395
  },
5343
- fetch: config?.fetch,
5396
+ fetch: fetchImpl,
5344
5397
  });
5345
5398
  if (models === null) {
5346
5399
  return null;
@@ -5704,6 +5757,15 @@ function openAiCompletionsDescriptor(
5704
5757
  return simpleModelsDevDescriptor(modelsDevKey, providerId, "openai-completions", baseUrl, options);
5705
5758
  }
5706
5759
 
5760
+ function openAiResponsesDescriptor(
5761
+ modelsDevKey: string,
5762
+ providerId: string,
5763
+ baseUrl: string,
5764
+ options: Omit<ModelsDevProviderDescriptor, "modelsDevKey" | "providerId" | "api" | "baseUrl"> = {},
5765
+ ): ModelsDevProviderDescriptor {
5766
+ return simpleModelsDevDescriptor(modelsDevKey, providerId, "openai-responses", baseUrl, options);
5767
+ }
5768
+
5707
5769
  function anthropicMessagesDescriptor(
5708
5770
  modelsDevKey: string,
5709
5771
  providerId: string,
@@ -5828,7 +5890,9 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
5828
5890
  defaultContextWindow: 131072,
5829
5891
  }),
5830
5892
  // --- xAI ---
5831
- openAiCompletionsDescriptor("xai", "xai", "https://api.x.ai/v1"),
5893
+ openAiResponsesDescriptor("xai", "xai", "https://api.x.ai/v1", {
5894
+ transformModel: model => applyXaiResponsesThinkingPolicy(model as ModelSpec<"openai-responses">),
5895
+ }),
5832
5896
  // --- DeepSeek ---
5833
5897
  openAiCompletionsDescriptor("deepseek", "deepseek", "https://api.deepseek.com", {
5834
5898
  // Only ship the v4 family as built-ins; older deepseek-chat / deepseek-reasoner
package/src/types.ts CHANGED
@@ -363,6 +363,13 @@ export interface OpenAICompat {
363
363
  * model id. Default: true. Issue #5606.
364
364
  */
365
365
  supportsSamplingParams?: boolean;
366
+ /**
367
+ * Whether presence/frequency penalties and stop sequences may be sent.
368
+ * First-party xAI `/v1/responses` rejects penalty fields for every model.
369
+ * xAI reasoning models also reject them (and `stop`) on chat completions.
370
+ * When unset, auto-detected. Default: true.
371
+ */
372
+ supportsPenaltyAndStopParams?: boolean;
366
373
  /** Always send a max-token field when the caller did not provide one. Default: auto-detected (Kimi-family models derive TPM limits from max_tokens). */
367
374
  alwaysSendMaxTokens?: boolean;
368
375
  /** Whether Responses-API tool-call/result history must be strictly paired. Default: auto-detected (Azure OpenAI, GitHub Copilot). */
@@ -578,6 +585,7 @@ export interface ResolvedOpenAISharedCompat {
578
585
  reasoningEffortMap: Partial<Record<Effort, string>>;
579
586
  supportsReasoningParams: boolean;
580
587
  supportsSamplingParams: boolean;
588
+ supportsPenaltyAndStopParams: boolean;
581
589
  thinkingFormat: OpenAIReasoningFormat;
582
590
  /** Kimi Code transport selected by live per-model protocol metadata. */
583
591
  kimiApiFormat?: OpenAICompat["kimiApiFormat"];
@@ -643,6 +651,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
643
651
  | "reasoningEffortMap"
644
652
  | "supportsReasoningParams"
645
653
  | "supportsSamplingParams"
654
+ | "supportsPenaltyAndStopParams"
646
655
  | "thinkingFormat"
647
656
  | "kimiApiFormat"
648
657
  | "reasoningDisableMode"
@@ -711,6 +720,12 @@ export interface ResolvedOpenAIResponsesCompat extends ResolvedOpenAISharedCompa
711
720
  strictResponsesPairing: boolean;
712
721
  supportsImageDetailOriginal: boolean;
713
722
  supportsObfuscationOptOut: boolean;
723
+ /**
724
+ * Whether `reasoning.summary` may be sent. First-party xAI `/v1/responses`
725
+ * rejects the field; handlers pass `null` so the wire omits it instead of
726
+ * filling `"auto"`.
727
+ */
728
+ supportsReasoningSummary: boolean;
714
729
  streamIdleTimeoutMs?: number;
715
730
  vercelGatewayRouting?: OpenAICompat["vercelGatewayRouting"];
716
731
  /** The model sits behind Vercel AI Gateway's Responses endpoint. */
package/src/wire/codex.ts CHANGED
@@ -11,6 +11,8 @@ export const CODEX_CLIENT_VERSION = "0.144.1";
11
11
 
12
12
  export const OPENAI_HEADERS = {
13
13
  BETA: "OpenAI-Beta",
14
+ /** Codex feature-negotiation header; values identify opt-in wire protocols. */
15
+ CODEX_BETA_FEATURES: "x-codex-beta-features",
14
16
  ACCOUNT_ID: "chatgpt-account-id",
15
17
  ORIGINATOR: "originator",
16
18
  VERSION: "version",
@@ -32,6 +34,7 @@ export const OPENAI_HEADERS = {
32
34
  export const OPENAI_HEADER_VALUES = {
33
35
  BETA_RESPONSES: "responses=experimental",
34
36
  BETA_RESPONSES_WEBSOCKETS_V2: "responses_websockets=2026-02-06",
37
+ REMOTE_COMPACTION_V2: "remote_compaction_v2",
35
38
  ORIGINATOR_CODEX: "pi",
36
39
  } as const;
37
40
 
@@ -1,3 +1,6 @@
1
+ import type { FetchImpl } from "../types";
2
+ import { isRecord } from "../utils";
3
+
1
4
  /**
2
5
  * GitHub Copilot wire metadata: API-key envelope parsing and endpoint
3
6
  * derivation shared by catalog discovery and the pi-ai OAuth flow. The device
@@ -72,6 +75,35 @@ export function normalizeGitHubCopilotApiEndpoint(input: string | undefined): st
72
75
  return undefined;
73
76
  }
74
77
  }
78
+ /**
79
+ * Resolve the plan-specific Copilot API endpoint advertised for a GitHub token.
80
+ * Login and raw environment-token discovery share this best-effort probe. Pass
81
+ * a `signal` to bound it against the same discovery deadline as `/models`; a
82
+ * stalled probe otherwise blocks discovery indefinitely.
83
+ */
84
+ export async function discoverGitHubCopilotApiEndpoint(
85
+ token: string,
86
+ fetchImpl: FetchImpl,
87
+ signal?: AbortSignal,
88
+ ): Promise<string | undefined> {
89
+ try {
90
+ const response = await fetchImpl("https://api.github.com/copilot_internal/user", {
91
+ headers: {
92
+ Accept: "application/json",
93
+ Authorization: `token ${token}`,
94
+ ...OPENCODE_HEADERS,
95
+ },
96
+ signal,
97
+ });
98
+ if (!response.ok) return undefined;
99
+ const data: unknown = await response.json();
100
+ if (!isRecord(data) || !isRecord(data.endpoints)) return undefined;
101
+ const endpoint = data.endpoints.api;
102
+ return typeof endpoint === "string" ? normalizeGitHubCopilotApiEndpoint(endpoint) : undefined;
103
+ } catch {
104
+ return undefined;
105
+ }
106
+ }
75
107
 
76
108
  export function parseGitHubCopilotApiKey(apiKeyRaw: string): ParsedGitHubCopilotApiKey {
77
109
  try {