@oh-my-pi/pi-catalog 17.3.4 → 17.3.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,29 @@
1
+ import type { TokenCost } from "./types";
2
+
3
+ /** Standard GPT-5.6 Sol rates used by the Daybreak Blue aliases. */
4
+ export const OPENAI_GPT_56_SOL_STANDARD_COST = {
5
+ input: 5,
6
+ output: 30,
7
+ cacheRead: 0.5,
8
+ cacheWrite: 6.25,
9
+ } as const satisfies TokenCost;
10
+
11
+ /** Standard GPT-5.6 Cyber rates used by the Daybreak Red aliases. */
12
+ export const OPENAI_GPT_56_CYBER_STANDARD_COST = {
13
+ input: 12.5,
14
+ output: 75,
15
+ cacheRead: 1.25,
16
+ cacheWrite: 15.625,
17
+ } as const satisfies TokenCost;
18
+
19
+ /** Resolve standard rates for Codex-prefixed Daybreak aliases. */
20
+ export function resolveOpenAIDaybreakStandardCost(modelId: string): TokenCost | undefined {
21
+ switch (modelId) {
22
+ case "gpt-daybreak-blue-latest":
23
+ return OPENAI_GPT_56_SOL_STANDARD_COST;
24
+ case "gpt-daybreak-red-latest":
25
+ return OPENAI_GPT_56_CYBER_STANDARD_COST;
26
+ default:
27
+ return undefined;
28
+ }
29
+ }
@@ -478,13 +478,13 @@ export const CATALOG_PROVIDERS = [
478
478
  },
479
479
  {
480
480
  id: "xai",
481
- defaultModel: "grok-4-fast-non-reasoning",
481
+ defaultModel: "grok-4.6",
482
482
  envVars: ["XAI_API_KEY"],
483
483
  createModelManagerOptions: (config: ModelManagerConfig) => xaiModelManagerOptions(config),
484
484
  },
485
485
  {
486
486
  id: "xai-oauth",
487
- defaultModel: "grok-4.3",
487
+ defaultModel: "grok-4.6",
488
488
  envVars: ["XAI_OAUTH_TOKEN", "XAI_API_KEY"],
489
489
  createModelManagerOptions: (config: ModelManagerConfig) => xaiOAuthModelManagerOptions(config),
490
490
  catalogDiscovery: {
@@ -522,7 +522,7 @@ export const CATALOG_PROVIDERS = [
522
522
  },
523
523
  {
524
524
  id: "zai",
525
- defaultModel: "glm-5.2",
525
+ defaultModel: "glm-5.3",
526
526
  envVars: ["ZAI_API_KEY"],
527
527
  createModelManagerOptions: (config: ModelManagerConfig) => zaiModelManagerOptions(config),
528
528
  catalogDiscovery: { label: "zAI" },
@@ -1,5 +1,6 @@
1
1
  import { USER_AGENT } from "@oh-my-pi/pi-utils";
2
2
  import * as logger from "@oh-my-pi/pi-utils/logger";
3
+ import { xaiResponsesReasoningEffortMap } from "../compat/openai";
3
4
  import {
4
5
  DEFAULT_OPENAI_COMPATIBLE_DISCOVERY_TIMEOUT_MS,
5
6
  fetchOpenAICompatibleModels,
@@ -20,7 +21,17 @@ import {
20
21
  import { resolveModelReference } from "../identity/reference";
21
22
  import type { ModelManagerOptions } from "../model-manager";
22
23
  import { type GeneratedProvider, getBundledModels } from "../models";
23
- import type { Api, FetchImpl, Model, ModelSpec, OpenAICompat, Provider, ThinkingConfig } from "../types";
24
+ import { OPENAI_GPT_56_CYBER_STANDARD_COST, OPENAI_GPT_56_SOL_STANDARD_COST } from "../openai-pricing";
25
+ import type {
26
+ Api,
27
+ FetchImpl,
28
+ LongContextTokenCost,
29
+ Model,
30
+ ModelSpec,
31
+ OpenAICompat,
32
+ Provider,
33
+ ThinkingConfig,
34
+ } from "../types";
24
35
  import { discoveryFetch, isAnthropicOAuthToken, isRecord, toBoolean, toNumber, toPositiveNumber } from "../utils";
25
36
  import { ALIBABA_TOKEN_PLAN_BASE_URL, parseAlibabaTokenPlanCredential } from "../wire/alibaba-token-plan";
26
37
  import { coreWeaveProjectHeaders } from "../wire/coreweave";
@@ -869,6 +880,7 @@ export function umansModelManagerOptions(config?: UmansModelManagerConfig): Mode
869
880
  // ---------------------------------------------------------------------------
870
881
 
871
882
  const OPENAI_API_BASE_URL = "https://api.openai.com/v1";
883
+ /** GPT-5.6 rates applied when a first-party request exceeds 272K input tokens. */
872
884
  export const OPENAI_GPT_56_LONG_CONTEXT_COSTS = {
873
885
  luna: {
874
886
  inputThreshold: 272_000,
@@ -891,20 +903,7 @@ export const OPENAI_GPT_56_LONG_CONTEXT_COSTS = {
891
903
  cacheRead: 0.4,
892
904
  cacheWrite: 5,
893
905
  },
894
- } as const;
895
- const OPENAI_GPT_56_SOL_STANDARD_COST = {
896
- input: 5,
897
- output: 30,
898
- cacheRead: 0.5,
899
- cacheWrite: 6.25,
900
- longContext: OPENAI_GPT_56_LONG_CONTEXT_COSTS.sol,
901
- } as const;
902
- const OPENAI_GPT_56_CYBER_STANDARD_COST = {
903
- input: 12.5,
904
- output: 75,
905
- cacheRead: 1.25,
906
- cacheWrite: 15.625,
907
- } as const;
906
+ } as const satisfies Readonly<Record<"luna" | "sol" | "terra", LongContextTokenCost>>;
908
907
 
909
908
  export interface OpenAIModelManagerConfig {
910
909
  apiKey?: string;
@@ -938,7 +937,10 @@ export const OPENAI_DAYBREAK_CURATED_FALLBACK_MODELS: readonly ModelSpec<"openai
938
937
  baseUrl: OPENAI_API_BASE_URL,
939
938
  reasoning: true,
940
939
  input: ["text", "image"],
941
- cost: OPENAI_GPT_56_SOL_STANDARD_COST,
940
+ cost: {
941
+ ...OPENAI_GPT_56_SOL_STANDARD_COST,
942
+ longContext: OPENAI_GPT_56_LONG_CONTEXT_COSTS.sol,
943
+ },
942
944
  contextWindow: 1_050_000,
943
945
  maxTokens: 128_000,
944
946
  },
@@ -1257,8 +1259,23 @@ export interface XaiModelManagerConfig {
1257
1259
  fetch?: FetchImpl;
1258
1260
  }
1259
1261
 
1260
- export function xaiModelManagerOptions(config?: XaiModelManagerConfig): ModelManagerOptions<"openai-completions"> {
1261
- return createSimpleOpenAICompletionsOptions("xai", "https://api.x.ai/v1", config);
1262
+ export function xaiModelManagerOptions(config?: XaiModelManagerConfig): ModelManagerOptions<"openai-responses"> {
1263
+ return {
1264
+ ...createOpenAICompatibleModelManagerOptions({
1265
+ api: "openai-responses",
1266
+ providerId: "xai",
1267
+ defaultBaseUrl: "https://api.x.ai/v1",
1268
+ config,
1269
+ requireApiKey: true,
1270
+ mapModel: mapWithBundledReference,
1271
+ }),
1272
+ // Completions → Responses migration: a fresh authoritative cache written
1273
+ // by the old resolver stores `api: "openai-completions"` for these ids.
1274
+ // Without a drop list, `online-if-uncached` skips the network and
1275
+ // `mergeDynamicModel` lets the cached api win over the new static
1276
+ // Responses entries until TTL expiry.
1277
+ dropCachedModelIdsOnStaticMismatch: getBundledModels("xai").map(model => model.id),
1278
+ };
1262
1279
  }
1263
1280
 
1264
1281
  export interface XaiOAuthModelManagerConfig {
@@ -1316,6 +1333,7 @@ export const XAI_OAUTH_CURATED_MODELS: readonly XAICuratedModel[] = [
1316
1333
  },
1317
1334
  { id: "grok-4.3", contextWindow: 1_000_000, name: "Grok 4.3", input: ["text", "image"] },
1318
1335
  { id: "grok-4.5", contextWindow: 500_000, name: "Grok 4.5", input: ["text", "image"] },
1336
+ { id: "grok-4.6", contextWindow: 500_000, name: "Grok 4.6", input: ["text", "image"] },
1319
1337
  // grok-4.20-multi-agent-0309 is text-only per the bundled catalog; omit `input` for the default.
1320
1338
  { id: "grok-4.20-multi-agent-0309", contextWindow: 2_000_000, name: "Grok 4.20 (Multi-Agent)" },
1321
1339
  {
@@ -1352,21 +1370,47 @@ const XAI_NON_CHAT_PREFIXES = ["grok-imagine-", "grok-stt-", "grok-voice-"] as c
1352
1370
  function withXaiOAuthCompatDefaults(model: ModelSpec<"openai-responses">): ModelSpec<"openai-responses"> {
1353
1371
  const compat = {
1354
1372
  ...(model.compat ?? {}),
1355
- includeEncryptedReasoning: model.compat?.includeEncryptedReasoning ?? false,
1356
- filterReasoningHistory: model.compat?.filterReasoningHistory ?? true,
1373
+ includeEncryptedReasoning: model.compat?.includeEncryptedReasoning ?? true,
1374
+ filterReasoningHistory: model.compat?.filterReasoningHistory ?? false,
1357
1375
  supportsImageDetailOriginal: model.compat?.supportsImageDetailOriginal ?? false,
1358
1376
  omitReasoningEffort: model.compat?.omitReasoningEffort ?? !isGrokReasoningEffortCapable(model.id),
1359
1377
  };
1360
1378
  return { ...model, compat };
1361
1379
  }
1362
1380
 
1363
- // Hermes-agent parity: only the `minimal -> low` clamp is applied (see
1364
- // hermes-agent/agent/transports/codex.py:92 `_effort_clamp = {"minimal":
1365
- // "low"}`). Hermes sends `xhigh` to xAI verbatim and we match that contract
1366
- // — let xAI decide if the level is valid for the specific Grok model.
1367
- // `resolveModelThinking` folds this into `model.thinking.effortMap`, downstream
1368
- // of the omitReasoningEffort gate in pi-ai's stream.ts.
1369
- const XAI_REASONING_EFFORT_MAP = { minimal: "low" } as const;
1381
+ // Hermes-agent parity for `minimal -> low` (see hermes-agent/agent/transports/
1382
+ // codex.py:92). Multi-agent Grok keeps `xhigh` unmapped (agent-count mode);
1383
+ // other first-party SKUs clamp leftover `xhigh`/`max` to `high`.
1384
+ // `resolveModelThinking` folds this into `model.thinking.effortMap`.
1385
+
1386
+ /**
1387
+ * Bake first-party xAI Responses effort-dial metadata onto a catalog spec.
1388
+ *
1389
+ * models.dev marks many Grok SKUs as reasoners and the thinking rebake would
1390
+ * otherwise emit a default `minimal/low/medium/high` dial. api.x.ai only
1391
+ * accepts `reasoning.effort` for {@link isGrokReasoningEffortCapable} ids —
1392
+ * off-allowlist reasoners (`grok-code-fast-1`, `grok-build-0.1`,
1393
+ * `grok-4.20-0309-reasoning`, …) 400 if the param is sent. SuperGrok
1394
+ * (`xai-oauth`) already curates this via {@link mergeCuratedIntoModel}; paid
1395
+ * `xai` rows come from stencil.so and need the same wire facts in the exported
1396
+ * `models.json` so direct catalog readers do not present an unsupported dial.
1397
+ *
1398
+ * Explicit `compat.supportsReasoningEffort` / `omitReasoningEffort` win.
1399
+ */
1400
+ export function applyXaiResponsesThinkingPolicy(model: ModelSpec<"openai-responses">): ModelSpec<"openai-responses"> {
1401
+ const effortCapable = model.compat?.supportsReasoningEffort ?? isGrokReasoningEffortCapable(model.id);
1402
+ const compat = {
1403
+ ...(model.compat ?? {}),
1404
+ supportsReasoningEffort: effortCapable,
1405
+ omitReasoningEffort: model.compat?.omitReasoningEffort ?? !effortCapable,
1406
+ };
1407
+ if (effortCapable) {
1408
+ compat.reasoningEffortMap = { ...xaiResponsesReasoningEffortMap(model.id) };
1409
+ } else {
1410
+ delete compat.reasoningEffortMap;
1411
+ }
1412
+ return { ...model, compat };
1413
+ }
1370
1414
 
1371
1415
  // xai-oauth's /v1/models exposes no per-request output limit on the OAuth
1372
1416
  // (Grok Build / SuperGrok) surface, so the curated catalog owns `maxTokens`
@@ -1381,9 +1425,9 @@ const XAI_REASONING_EFFORT_MAP = { minimal: "low" } as const;
1381
1425
  // reasoning metadata and fetchOpenAICompatibleModels defaults reasoning to
1382
1426
  // false). Caller supplies a `base` Model (either a freshly synthesised seed
1383
1427
  // or a dynamic-fetched entry); the helper layers curated fields on top.
1384
- // The `minimal -> low` effort clamp (XAI_REASONING_EFFORT_MAP) is always
1385
- // merged in so dynamic-fetched models — which arrive without curated
1386
- // compat keys — still get the clamp applyResponsesReasoningParams expects.
1428
+ // The effort remap from {@link xaiResponsesReasoningEffortMap} is merged
1429
+ // only onto effort-capable rows. Off-allowlist reasoners omit the wire
1430
+ // param, so a map on those specs is dead weight.
1387
1431
  // The effort-dial pair (`supportsReasoningEffort`/`omitReasoningEffort`) is
1388
1432
  // authoritative: a stale flag on `base` (previous snapshot or dynamic fetch)
1389
1433
  // must not outlive an allowlist change in identity/family.ts.
@@ -1394,13 +1438,17 @@ function mergeCuratedIntoModel(
1394
1438
  const effortCapable = curated.supportsReasoningEffort ?? isGrokReasoningEffortCapable(curated.id);
1395
1439
  const compat = {
1396
1440
  ...(base.compat ?? {}),
1397
- reasoningEffortMap: { ...XAI_REASONING_EFFORT_MAP, ...(base.compat?.reasoningEffortMap ?? {}) },
1398
- includeEncryptedReasoning: base.compat?.includeEncryptedReasoning ?? false,
1399
- filterReasoningHistory: base.compat?.filterReasoningHistory ?? true,
1441
+ includeEncryptedReasoning: base.compat?.includeEncryptedReasoning ?? true,
1442
+ filterReasoningHistory: false,
1400
1443
  supportsImageDetailOriginal: base.compat?.supportsImageDetailOriginal ?? false,
1401
1444
  omitReasoningEffort: !effortCapable,
1402
1445
  supportsReasoningEffort: effortCapable,
1403
1446
  };
1447
+ if (effortCapable) {
1448
+ compat.reasoningEffortMap = { ...xaiResponsesReasoningEffortMap(curated.id) };
1449
+ } else {
1450
+ delete compat.reasoningEffortMap;
1451
+ }
1404
1452
  return {
1405
1453
  ...base,
1406
1454
  contextWindow: curated.contextWindow,
@@ -1502,7 +1550,7 @@ export function buildXaiOAuthStaticSeed(baseUrl?: string): ModelSpec<"openai-res
1502
1550
  cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
1503
1551
  contextWindow: curated.contextWindow,
1504
1552
  maxTokens: curated.contextWindow,
1505
- compat: { reasoningEffortMap: XAI_REASONING_EFFORT_MAP },
1553
+ compat: { reasoningEffortMap: xaiResponsesReasoningEffortMap(curated.id) },
1506
1554
  };
1507
1555
  return mergeCuratedIntoModel(base, curated);
1508
1556
  });
@@ -3634,15 +3682,19 @@ export function basetenModelManagerOptions(
3634
3682
  const features = Array.isArray(raw.supported_features) ? raw.supported_features : [];
3635
3683
  const modalities = Array.isArray(raw.input_modalities) ? raw.input_modalities : [];
3636
3684
 
3637
- // Baseten's reasoning router accepts only the high/max
3638
- // effort tiers for its GLM-5.2 and gpt-oss routes.
3639
- const isEffortReasoning =
3685
+ // Baseten's discovery flags are not enough to enable OMP reasoning for every
3686
+ // model. Only models with a verified Baseten reasoning policy are enabled
3687
+ // here; an unknown model may use a different reasoning wire shape or effort
3688
+ // vocabulary, which OMP must not guess.
3689
+ const isSupportedBasetenReasoningModel =
3690
+ isKimiK3ModelId(defaults.id) ||
3640
3691
  defaults.id === "openai/gpt-oss-120b" ||
3692
+ defaults.id === "deepseek-ai/DeepSeek-V4-Pro" ||
3641
3693
  defaults.id === "zai-org/GLM-5.2" ||
3642
3694
  defaults.id === "zai-org/GLM-5.2-Fast";
3643
- const isBasetenNativeReasoning = isEffortReasoning || defaults.id === "deepseek-ai/DeepSeek-V4-Pro";
3644
3695
  const reasoning =
3645
- isBasetenNativeReasoning && (features.includes("reasoning") || features.includes("reasoning_effort"));
3696
+ isSupportedBasetenReasoningModel &&
3697
+ (features.includes("reasoning") || features.includes("reasoning_effort"));
3646
3698
  const supportsTools = features.includes("tools") ? undefined : false;
3647
3699
  const vision = modalities.includes("image") || (reference?.input.includes("image") ?? false);
3648
3700
 
@@ -3656,14 +3708,7 @@ export function basetenModelManagerOptions(
3656
3708
 
3657
3709
  const contextWindow = toPositiveNumber(raw.context_length, reference?.contextWindow ?? defaults.contextWindow);
3658
3710
  const maxTokens = toPositiveNumber(raw.max_completion_tokens, reference?.maxTokens ?? defaults.maxTokens);
3659
-
3660
3711
  const baseModel = mapWithBundledReference(entry, defaults, reference);
3661
- const thinking = isEffortReasoning
3662
- ? {
3663
- mode: "effort" as const,
3664
- efforts: [Effort.High, Effort.Max],
3665
- }
3666
- : undefined;
3667
3712
 
3668
3713
  return {
3669
3714
  ...baseModel,
@@ -3672,7 +3717,6 @@ export function basetenModelManagerOptions(
3672
3717
  cost,
3673
3718
  contextWindow,
3674
3719
  maxTokens,
3675
- ...(thinking ? { thinking } : {}),
3676
3720
  ...(supportsTools === false ? { supportsTools } : {}),
3677
3721
  };
3678
3722
  },
@@ -5713,6 +5757,15 @@ function openAiCompletionsDescriptor(
5713
5757
  return simpleModelsDevDescriptor(modelsDevKey, providerId, "openai-completions", baseUrl, options);
5714
5758
  }
5715
5759
 
5760
+ function openAiResponsesDescriptor(
5761
+ modelsDevKey: string,
5762
+ providerId: string,
5763
+ baseUrl: string,
5764
+ options: Omit<ModelsDevProviderDescriptor, "modelsDevKey" | "providerId" | "api" | "baseUrl"> = {},
5765
+ ): ModelsDevProviderDescriptor {
5766
+ return simpleModelsDevDescriptor(modelsDevKey, providerId, "openai-responses", baseUrl, options);
5767
+ }
5768
+
5716
5769
  function anthropicMessagesDescriptor(
5717
5770
  modelsDevKey: string,
5718
5771
  providerId: string,
@@ -5837,7 +5890,9 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
5837
5890
  defaultContextWindow: 131072,
5838
5891
  }),
5839
5892
  // --- xAI ---
5840
- openAiCompletionsDescriptor("xai", "xai", "https://api.x.ai/v1"),
5893
+ openAiResponsesDescriptor("xai", "xai", "https://api.x.ai/v1", {
5894
+ transformModel: model => applyXaiResponsesThinkingPolicy(model as ModelSpec<"openai-responses">),
5895
+ }),
5841
5896
  // --- DeepSeek ---
5842
5897
  openAiCompletionsDescriptor("deepseek", "deepseek", "https://api.deepseek.com", {
5843
5898
  // Only ship the v4 family as built-ins; older deepseek-chat / deepseek-reasoner
package/src/types.ts CHANGED
@@ -363,6 +363,13 @@ export interface OpenAICompat {
363
363
  * model id. Default: true. Issue #5606.
364
364
  */
365
365
  supportsSamplingParams?: boolean;
366
+ /**
367
+ * Whether presence/frequency penalties and stop sequences may be sent.
368
+ * First-party xAI `/v1/responses` rejects penalty fields for every model.
369
+ * xAI reasoning models also reject them (and `stop`) on chat completions.
370
+ * When unset, auto-detected. Default: true.
371
+ */
372
+ supportsPenaltyAndStopParams?: boolean;
366
373
  /** Always send a max-token field when the caller did not provide one. Default: auto-detected (Kimi-family models derive TPM limits from max_tokens). */
367
374
  alwaysSendMaxTokens?: boolean;
368
375
  /** Whether Responses-API tool-call/result history must be strictly paired. Default: auto-detected (Azure OpenAI, GitHub Copilot). */
@@ -578,6 +585,7 @@ export interface ResolvedOpenAISharedCompat {
578
585
  reasoningEffortMap: Partial<Record<Effort, string>>;
579
586
  supportsReasoningParams: boolean;
580
587
  supportsSamplingParams: boolean;
588
+ supportsPenaltyAndStopParams: boolean;
581
589
  thinkingFormat: OpenAIReasoningFormat;
582
590
  /** Kimi Code transport selected by live per-model protocol metadata. */
583
591
  kimiApiFormat?: OpenAICompat["kimiApiFormat"];
@@ -643,6 +651,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
643
651
  | "reasoningEffortMap"
644
652
  | "supportsReasoningParams"
645
653
  | "supportsSamplingParams"
654
+ | "supportsPenaltyAndStopParams"
646
655
  | "thinkingFormat"
647
656
  | "kimiApiFormat"
648
657
  | "reasoningDisableMode"
@@ -711,6 +720,12 @@ export interface ResolvedOpenAIResponsesCompat extends ResolvedOpenAISharedCompa
711
720
  strictResponsesPairing: boolean;
712
721
  supportsImageDetailOriginal: boolean;
713
722
  supportsObfuscationOptOut: boolean;
723
+ /**
724
+ * Whether `reasoning.summary` may be sent. First-party xAI `/v1/responses`
725
+ * rejects the field; handlers pass `null` so the wire omits it instead of
726
+ * filling `"auto"`.
727
+ */
728
+ supportsReasoningSummary: boolean;
714
729
  streamIdleTimeoutMs?: number;
715
730
  vercelGatewayRouting?: OpenAICompat["vercelGatewayRouting"];
716
731
  /** The model sits behind Vercel AI Gateway's Responses endpoint. */