@oh-my-pi/pi-catalog 17.0.0 → 17.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -8,6 +8,7 @@ import { FIREWORKS_FAST_SUFFIX, toFireworksPublicModelId } from "../fireworks-mo
8
8
  import {
9
9
  isGlmVisionModelId,
10
10
  isGrokReasoningEffortCapable,
11
+ isKimiK3ModelId,
11
12
  isKimiModelId,
12
13
  isReasoningGlmModelId,
13
14
  } from "../identity/family";
@@ -1494,13 +1495,31 @@ export function zhipuCodingPlanModelManagerOptions(
1494
1495
  export const FIREWORKS_KIMI_MAX_TOKENS = 32_768;
1495
1496
 
1496
1497
  /**
1497
- * Returns true for any Kimi K2.x public model id served by Fireworks-backed
1498
- * providers (`fireworks` direct, `firepass` router). Matches both the public
1499
- * catalog id (`kimi-k2.5`, `kimi-k2.6`, `kimi-k2.6-turbo`) and the canonical
1500
- * Fireworks wire id (`accounts/fireworks/{models,routers}/kimi-k2…`).
1498
+ * Fireworks' output ceiling for Kimi K2.7-Code specifically. Its `/v1/models`
1499
+ * generic `max_completion_tokens` is 65,536 and Fireworks serves it in full —
1500
+ * verified with a single completion emitting 58,971 output tokens and
1501
+ * `max_tokens: 200000` accepted without error. Unlike the older K2.5/K2.6
1502
+ * family (see {@link FIREWORKS_KIMI_MAX_TOKENS}), K2.7-Code is not clamped to
1503
+ * 32,768; that ceiling only truncated it.
1504
+ */
1505
+ export const FIREWORKS_KIMI_K27_CODE_MAX_TOKENS = 65_536;
1506
+
1507
+ /**
1508
+ * Returns true for the Kimi K2.5 / K2.6 family served by Fireworks-backed
1509
+ * providers (`fireworks` direct, `firepass` router) that share the 32,768
1510
+ * `maxTokens` ceiling. Matches both the public catalog id (`kimi-k2.5`,
1511
+ * `kimi-k2.6`, `kimi-k2.6-turbo`) and the canonical Fireworks wire id
1512
+ * (`accounts/fireworks/{models,routers}/kimi-k2…`).
1513
+ *
1514
+ * K2.7-Code (incl. `-fast` / `-highspeed`) is deliberately excluded: unlike the
1515
+ * earlier K2 family it serves its full context on Fireworks — verified with a
1516
+ * single completion emitting 58,971 output tokens and `max_tokens: 200000`
1517
+ * accepted without error — so the 32,768 cap would only truncate it. It inherits
1518
+ * Fireworks' reported `max_completion_tokens` (65,536) instead.
1501
1519
  */
1502
1520
  export function isFireworksKimiK2ModelId(modelId: string): boolean {
1503
1521
  const trimmed = modelId.toLowerCase();
1522
+ if (/kimi[-._]?k2(?:[._-]?|p)7[-._]?code/.test(trimmed)) return false;
1504
1523
  if (trimmed.startsWith("kimi-k2")) return true;
1505
1524
  return /\/kimi-k2(?:p\d+)?(?:[._-]|$)/.test(trimmed);
1506
1525
  }
@@ -1657,9 +1676,14 @@ function mapFireworksControlPlaneModel(
1657
1676
  const supportsImage = toBoolean(record.supportsImageInput) === true;
1658
1677
  const supportsTools = toBoolean(record.supportsTools);
1659
1678
  const contextWindow = toPositiveNumber(record.contextLength, reference?.contextWindow ?? null);
1660
- // The control plane reports no max-output budget; default the Kimi family to
1661
- // its published cap, everyone else to the discovery fallback, then clamp.
1662
- const fallbackMaxTokens = isFireworksKimiK2ModelId(publicModelId) ? FIREWORKS_KIMI_MAX_TOKENS : null;
1679
+ // The control plane reports no max-output budget. Default K2.7-Code to its
1680
+ // verified 65,536 ceiling, the older K2.5/K2.6 family to the clamped 32,768,
1681
+ // everyone else to the discovery fallback, then clamp.
1682
+ const fallbackMaxTokens = isKimiK27CodeModelId(publicModelId)
1683
+ ? FIREWORKS_KIMI_K27_CODE_MAX_TOKENS
1684
+ : isFireworksKimiK2ModelId(publicModelId)
1685
+ ? FIREWORKS_KIMI_MAX_TOKENS
1686
+ : null;
1663
1687
  const maxTokens = clampFireworksKimiMaxTokens(publicModelId, reference?.maxTokens ?? fallbackMaxTokens);
1664
1688
  const base: ModelSpec<"openai-completions"> = reference ?? {
1665
1689
  id: publicModelId,
@@ -2893,6 +2917,23 @@ export interface MoonshotModelManagerConfig {
2893
2917
  fetch?: FetchImpl;
2894
2918
  }
2895
2919
 
2920
+ /**
2921
+ * Moonshot Kimi K3 discovery metadata. K3 is dynamically discovered but absent
2922
+ * from models.dev and the bundled catalog, so `mapWithBundledReference` would
2923
+ * otherwise assign zero cost, null limits, text-only input, and no reasoning —
2924
+ * mislabeling a paid model as "Free" (#5756). Pricing/limits from Moonshot's
2925
+ * official chat-k3 pricing and quickstart guide:
2926
+ * https://platform.kimi.ai/docs/pricing/chat-k3.md
2927
+ * https://platform.kimi.ai/docs/guide/kimi-k3-quickstart
2928
+ * K3 always reasons and supports only `reasoning_effort: "max"` — it does NOT
2929
+ * use the K2.x binary `thinking: { type }` block, so the wire path routes it
2930
+ * through OpenAI-style `reasoning_effort` (see `buildOpenAICompat`).
2931
+ */
2932
+ const MOONSHOT_KIMI_K3_COST = { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 0 } as const;
2933
+ const MOONSHOT_KIMI_K3_CONTEXT_WINDOW = 1_048_576;
2934
+ const MOONSHOT_KIMI_K3_MAX_TOKENS = 131_072;
2935
+ const MOONSHOT_KIMI_K3_THINKING: ThinkingConfig = { mode: "effort", efforts: [Effort.Max], requiresEffort: true };
2936
+
2896
2937
  export function moonshotModelManagerOptions(
2897
2938
  config?: MoonshotModelManagerConfig,
2898
2939
  ): ModelManagerOptions<"openai-completions"> {
@@ -2915,6 +2956,25 @@ export function moonshotModelManagerOptions(
2915
2956
  const reference = references.get(defaults.id);
2916
2957
  const model = mapWithBundledReference(entry, defaults, reference);
2917
2958
  const id = model.id.toLowerCase();
2959
+ // Kimi K3 is discovered but has no bundled/models.dev reference, so the
2960
+ // generic dynamic defaults would report it "Free" with no capabilities
2961
+ // (#5756). Stamp the official pricing/limits when the endpoint doesn't
2962
+ // carry them, and mark it reasoning + vision. K3 always reasons via
2963
+ // `reasoning_effort: "max"` and does NOT use the K2.x `thinking` block,
2964
+ // so its thinking config is the single-tier `max` scale — the wire path
2965
+ // routes it through `reasoning_effort` (see `buildOpenAICompat`).
2966
+ if (!reference && isKimiK3ModelId(id)) {
2967
+ const isZeroCost = model.cost.input === 0 && model.cost.output === 0 && model.cost.cacheRead === 0;
2968
+ return {
2969
+ ...model,
2970
+ reasoning: true,
2971
+ input: ["text", "image"],
2972
+ cost: isZeroCost ? { ...MOONSHOT_KIMI_K3_COST } : model.cost,
2973
+ contextWindow: model.contextWindow ?? MOONSHOT_KIMI_K3_CONTEXT_WINDOW,
2974
+ maxTokens: model.maxTokens ?? MOONSHOT_KIMI_K3_MAX_TOKENS,
2975
+ thinking: model.thinking ?? { ...MOONSHOT_KIMI_K3_THINKING },
2976
+ };
2977
+ }
2918
2978
  // Moonshot's K2.x family (K2.5, K2.6, kimi-k2-thinking, …) is reasoning-capable
2919
2979
  // and vision-capable on the native API. Without these flags the openai-completions
2920
2980
  // path skips the z.ai-format `thinking` block, and Moonshot K2.6 stalls on first
@@ -3724,7 +3784,8 @@ export interface GithubCopilotModelManagerConfig {
3724
3784
 
3725
3785
  const COPILOT_ANTHROPIC_MODEL_PATTERN = /^claude-(haiku|sonnet|opus|fable|mythos)-\d/;
3726
3786
  const isCopilotResponsesModelId = (modelId: string): boolean =>
3727
- modelId.startsWith("gpt-5") || modelId.startsWith("oswe");
3787
+ modelId.startsWith("gpt-5") || modelId.startsWith("oswe") || modelId.startsWith("mai-");
3788
+ const COPILOT_CACHE_INVALIDATED_MODEL_IDS = ["mai-code-1-flash-picker"];
3728
3789
 
3729
3790
  function inferCopilotApi(modelId: string): Api {
3730
3791
  if (COPILOT_ANTHROPIC_MODEL_PATTERN.test(modelId)) {
@@ -3888,6 +3949,7 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
3888
3949
  const resolveReference = createReferenceResolver(providerRefs);
3889
3950
  return {
3890
3951
  providerId: "github-copilot",
3952
+ dropCachedModelIdsOnStaticMismatch: COPILOT_CACHE_INVALIDATED_MODEL_IDS,
3891
3953
  ...(apiKey && {
3892
3954
  fetchDynamicModels: async () => {
3893
3955
  const longContextVariants: ModelSpec<Api>[] = [];
@@ -4505,9 +4567,17 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
4505
4567
 
4506
4568
  const MODELS_DEV_PROVIDER_DESCRIPTORS_CODING_PLANS: readonly ModelsDevProviderDescriptor[] = [
4507
4569
  // --- zAI ---
4508
- anthropicMessagesDescriptor("zai-coding-plan", "zai", "https://api.z.ai/api/anthropic"),
4509
- // --- Umans AI Coding Plan ---
4510
- anthropicMessagesDescriptor("umans-ai-coding-plan", "umans", UMANS_BASE_URL),
4570
+ // Source the models.dev `zai` (pay-as-you-go) key rather than `zai-coding-plan`:
4571
+ // the coding-plan key reports all-$0 subscription rates, which surface every GLM
4572
+ // SKU as "Free" in `/models`. The PAYG key carries the real per-token rates for
4573
+ // the identical model ids, so the enumerated token costs line up with the other
4574
+ // subscription providers for comparison (issue #5598).
4575
+ anthropicMessagesDescriptor("zai", "zai", "https://api.z.ai/api/anthropic"),
4576
+ // --- Umans AI ---
4577
+ // Source the pay-as-you-go catalog: the coding-plan key publishes subscription
4578
+ // costs as zero, while `/models/info` omits pricing entirely. The generator
4579
+ // overlays these rates onto the authoritative endpoint discovery (issue #5733).
4580
+ anthropicMessagesDescriptor("umans-ai", "umans", UMANS_BASE_URL),
4511
4581
  // --- Xiaomi ---
4512
4582
  openAiCompletionsDescriptor("xiaomi", "xiaomi", "https://api.xiaomimimo.com/v1", {
4513
4583
  defaultContextWindow: 262144,
package/src/types.ts CHANGED
@@ -327,6 +327,15 @@ export interface OpenAICompat {
327
327
  toolStrictMode?: "all_strict" | "none";
328
328
  /** Whether request shaping may send reasoning params at all. Default: auto-detected (disabled for GitHub Copilot chat-completions). */
329
329
  supportsReasoningParams?: boolean;
330
+ /**
331
+ * Whether the endpoint accepts explicit sampling parameters (`temperature`,
332
+ * `top_p`, `top_k`, `min_p`, penalties). OpenAI proprietary reasoning models
333
+ * (o-series, gpt-5+) reject them with `400 Unsupported parameter:
334
+ * 'temperature' is not supported with this model` on every serving host
335
+ * (official, Azure, GitHub Copilot). When unset, auto-detected from the
336
+ * model id. Default: true. Issue #5606.
337
+ */
338
+ supportsSamplingParams?: boolean;
330
339
  /** Always send a max-token field when the caller did not provide one. Default: auto-detected (Kimi-family models derive TPM limits from max_tokens). */
331
340
  alwaysSendMaxTokens?: boolean;
332
341
  /** Whether Responses-API tool-call/result history must be strictly paired. Default: auto-detected (Azure OpenAI, GitHub Copilot). */
@@ -372,7 +381,7 @@ export interface AnthropicCompat {
372
381
  * tags: 'disabled', 'enabled'`.
373
382
  */
374
383
  disableAdaptiveThinking?: boolean;
375
- /** Whether tools may include Anthropic's per-tool eager_input_streaming flag. Default: true. */
384
+ /** Whether tools may include Anthropic's per-tool eager_input_streaming flag. Default: true for the canonical Anthropic API. */
376
385
  supportsEagerToolInputStreaming?: boolean;
377
386
  /** Whether long prompt-cache retention (`ttl: "1h"`) is supported. Default: true for canonical Anthropic API. */
378
387
  supportsLongCacheRetention?: boolean;
@@ -464,6 +473,7 @@ export interface ResolvedOpenAISharedCompat {
464
473
  supportsReasoningEffort: boolean;
465
474
  reasoningEffortMap: Partial<Record<Effort, string>>;
466
475
  supportsReasoningParams: boolean;
476
+ supportsSamplingParams: boolean;
467
477
  thinkingFormat: OpenAIReasoningFormat;
468
478
  reasoningDisableMode: OpenAIReasoningDisableMode;
469
479
  omitReasoningEffort: boolean;
@@ -516,6 +526,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
516
526
  | "supportsReasoningEffort"
517
527
  | "reasoningEffortMap"
518
528
  | "supportsReasoningParams"
529
+ | "supportsSamplingParams"
519
530
  | "thinkingFormat"
520
531
  | "reasoningDisableMode"
521
532
  | "omitReasoningEffort"