@oh-my-pi/pi-catalog 17.3.7 → 17.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/CHANGELOG.md +36 -0
  2. package/dist/types/build.d.ts +5 -0
  3. package/dist/types/discovery/cursor-proto.d.ts +5362 -0
  4. package/dist/types/discovery/devin-proto.d.ts +1549 -0
  5. package/dist/types/discovery/index.d.ts +1 -0
  6. package/dist/types/discovery/protobuf.d.ts +102 -0
  7. package/dist/types/identity/family.d.ts +23 -1
  8. package/dist/types/index.d.ts +1 -0
  9. package/dist/types/model-thinking.d.ts +6 -2
  10. package/dist/types/model-tokenizer.d.ts +8 -0
  11. package/dist/types/provider-models/descriptors.d.ts +1 -0
  12. package/dist/types/types.d.ts +27 -1
  13. package/dist/types/variant-collapse.d.ts +2 -0
  14. package/package.json +5 -6
  15. package/src/build.ts +7 -0
  16. package/src/compat/openai.ts +27 -1
  17. package/src/discovery/codex.ts +118 -24
  18. package/src/discovery/cursor-proto.ts +8322 -0
  19. package/src/discovery/cursor.ts +16 -3
  20. package/src/discovery/devin-proto.ts +2037 -0
  21. package/src/discovery/devin.ts +4 -3
  22. package/src/discovery/index.ts +1 -0
  23. package/src/discovery/protobuf.ts +1073 -0
  24. package/src/hosts.ts +22 -1
  25. package/src/identity/family.ts +42 -1
  26. package/src/index.ts +1 -0
  27. package/src/model-cache.ts +67 -6
  28. package/src/model-thinking.ts +38 -6
  29. package/src/model-tokenizer.ts +75 -0
  30. package/src/models.json +185055 -2118
  31. package/src/models.ts +6 -5
  32. package/src/provider-models/bundled-references.ts +12 -2
  33. package/src/provider-models/cache-provider-id.ts +3 -1
  34. package/src/provider-models/descriptors.ts +1 -0
  35. package/src/provider-models/openai-compat.ts +194 -44
  36. package/src/types.ts +36 -0
  37. package/src/variant-collapse.ts +67 -34
  38. package/dist/types/discovery/cursor-gen/agent_pb.d.ts +0 -16769
  39. package/dist/types/discovery/devin-gen/buf/validate/validate_pb.d.ts +0 -1715
  40. package/dist/types/discovery/devin-gen/exa/analytics_pb/analytics_pb.d.ts +0 -693
  41. package/dist/types/discovery/devin-gen/exa/api_server_pb/api_server_pb.d.ts +0 -9158
  42. package/dist/types/discovery/devin-gen/exa/auth_pb/auth_pb.d.ts +0 -52
  43. package/dist/types/discovery/devin-gen/exa/auto_cascade_common_pb/auto_cascade_common_pb.d.ts +0 -276
  44. package/dist/types/discovery/devin-gen/exa/bug_checker_pb/bug_checker_pb.d.ts +0 -78
  45. package/dist/types/discovery/devin-gen/exa/cascade_plugins_pb/cascade_plugins_pb.d.ts +0 -999
  46. package/dist/types/discovery/devin-gen/exa/chat_pb/chat_pb.d.ts +0 -1704
  47. package/dist/types/discovery/devin-gen/exa/code_edit/code_edit_pb/code_edit_pb.d.ts +0 -656
  48. package/dist/types/discovery/devin-gen/exa/codeium_common_pb/codeium_common_pb.d.ts +0 -15014
  49. package/dist/types/discovery/devin-gen/exa/context_module_pb/context_module_pb.d.ts +0 -607
  50. package/dist/types/discovery/devin-gen/exa/cortex_pb/cortex_pb.d.ts +0 -11830
  51. package/dist/types/discovery/devin-gen/exa/diff_action_pb/diff_action_pb.d.ts +0 -248
  52. package/dist/types/discovery/devin-gen/exa/index_pb/index_pb.d.ts +0 -1747
  53. package/dist/types/discovery/devin-gen/exa/knowledge_base_pb/knowledge_base_pb.d.ts +0 -509
  54. package/dist/types/discovery/devin-gen/exa/language_server_pb/language_server_pb.d.ts +0 -9048
  55. package/dist/types/discovery/devin-gen/exa/opensearch_clients_pb/opensearch_clients_pb.d.ts +0 -1760
  56. package/dist/types/discovery/devin-gen/exa/prompt_pb/prompt_pb.d.ts +0 -286
  57. package/dist/types/discovery/devin-gen/exa/reactive_component_pb/reactive_component_pb.d.ts +0 -405
  58. package/dist/types/discovery/devin-gen/exa/trust_pb/trust_pb.d.ts +0 -582
  59. package/src/discovery/cursor-gen/agent_pb.ts +0 -19513
  60. package/src/discovery/devin-gen/buf/validate/validate_pb.ts +0 -1862
  61. package/src/discovery/devin-gen/exa/analytics_pb/analytics_pb.ts +0 -871
  62. package/src/discovery/devin-gen/exa/api_server_pb/api_server_pb.ts +0 -11083
  63. package/src/discovery/devin-gen/exa/auth_pb/auth_pb.ts +0 -71
  64. package/src/discovery/devin-gen/exa/auto_cascade_common_pb/auto_cascade_common_pb.ts +0 -348
  65. package/src/discovery/devin-gen/exa/bug_checker_pb/bug_checker_pb.ts +0 -103
  66. package/src/discovery/devin-gen/exa/cascade_plugins_pb/cascade_plugins_pb.ts +0 -1198
  67. package/src/discovery/devin-gen/exa/chat_pb/chat_pb.ts +0 -2063
  68. package/src/discovery/devin-gen/exa/code_edit/code_edit_pb/code_edit_pb.ts +0 -810
  69. package/src/discovery/devin-gen/exa/codeium_common_pb/codeium_common_pb.ts +0 -18354
  70. package/src/discovery/devin-gen/exa/context_module_pb/context_module_pb.ts +0 -732
  71. package/src/discovery/devin-gen/exa/cortex_pb/cortex_pb.ts +0 -14277
  72. package/src/discovery/devin-gen/exa/diff_action_pb/diff_action_pb.ts +0 -312
  73. package/src/discovery/devin-gen/exa/index_pb/index_pb.ts +0 -2106
  74. package/src/discovery/devin-gen/exa/knowledge_base_pb/knowledge_base_pb.ts +0 -623
  75. package/src/discovery/devin-gen/exa/language_server_pb/language_server_pb.ts +0 -10918
  76. package/src/discovery/devin-gen/exa/opensearch_clients_pb/opensearch_clients_pb.ts +0 -2125
  77. package/src/discovery/devin-gen/exa/prompt_pb/prompt_pb.ts +0 -361
  78. package/src/discovery/devin-gen/exa/reactive_component_pb/reactive_component_pb.ts +0 -450
  79. package/src/discovery/devin-gen/exa/trust_pb/trust_pb.ts +0 -712
@@ -3,3 +3,4 @@ export * from "./codex.js";
3
3
  export * from "./gemini.js";
4
4
  export * from "./gitlab-duo-workflow.js";
5
5
  export * from "./openai-compatible.js";
6
+ export * from "./protobuf.js";
@@ -0,0 +1,102 @@
1
+ /**
2
+ * High-performance, zero-builder protobuf wire codecs for @oh-my-pi/pi-catalog.
3
+ *
4
+ * Schemas are declared as static IR descriptors with near-zero module load overhead
5
+ * and lazy compilation on first encode/decode/create invocation.
6
+ */
7
+ /** JSON values carried by `google.protobuf.Value` fields. */
8
+ export type JsonValue = null | boolean | number | string | JsonValue[] | {
9
+ [key: string]: JsonValue;
10
+ };
11
+ /** An unrecognised wire field retained for forward-compatible round-trips. */
12
+ export interface ProtoUnknownField {
13
+ no: number;
14
+ wireType: number;
15
+ data: Uint8Array;
16
+ }
17
+ /** Shared internal metadata present on every decoded protocol message. */
18
+ export interface ProtoMessage {
19
+ $typeName?: string;
20
+ $unknown?: ProtoUnknownField[];
21
+ }
22
+ /** A bidirectional codec for one protobuf message type. */
23
+ export interface MessageCodec<T extends ProtoMessage = ProtoMessage> {
24
+ (value: T): Uint8Array;
25
+ (value: Uint8Array): T;
26
+ /** Creates a message with protobuf defaults for omitted fields. */
27
+ create(value?: Partial<T>): T;
28
+ /** Encodes one message into protobuf wire bytes. */
29
+ encode(value: T): Uint8Array;
30
+ /** Decodes one protobuf message from wire bytes. */
31
+ decode(value: Uint8Array): T;
32
+ /** Converts a message to its protobuf JSON representation. */
33
+ toJson(value: T): JsonValue;
34
+ }
35
+ /** Infers a message shape from a codec result. */
36
+ export type InferMessage<TCodec> = TCodec extends MessageCodec<infer TMessage> ? TMessage : never;
37
+ /** Erases a referenced message's concrete shape for static field descriptors. */
38
+ export interface MessageReference {
39
+ encode(value: unknown): Uint8Array;
40
+ decode(value: Uint8Array): ProtoMessage;
41
+ toJson(value: unknown): JsonValue;
42
+ }
43
+ export type ScalarKind = "bool" | "bytes" | "double" | "enum" | "float" | "int32" | "int64" | "string" | "uint32" | "uint64";
44
+ export type WireType = 0 | 1 | 2 | 5;
45
+ export interface ScalarFieldDesc {
46
+ readonly no: number;
47
+ readonly name: string;
48
+ readonly kind: ScalarKind;
49
+ readonly optional?: boolean;
50
+ readonly repeat?: boolean;
51
+ }
52
+ export interface MessageFieldDesc {
53
+ readonly no: number;
54
+ readonly name: string;
55
+ readonly kind: "message";
56
+ readonly T: () => MessageReference;
57
+ readonly repeat?: boolean;
58
+ }
59
+ export interface EnumFieldDesc {
60
+ readonly no: number;
61
+ readonly name: string;
62
+ readonly kind: "enum";
63
+ readonly optional?: boolean;
64
+ readonly repeat?: boolean;
65
+ }
66
+ export interface MapFieldDesc {
67
+ readonly no: number;
68
+ readonly name: string;
69
+ readonly kind: "map";
70
+ readonly K: "string";
71
+ readonly V: ScalarKind | (() => MessageReference);
72
+ }
73
+ export type VariantDesc = {
74
+ readonly no: number;
75
+ readonly name: string;
76
+ readonly kind: ScalarKind;
77
+ } | {
78
+ readonly no: number;
79
+ readonly name: string;
80
+ readonly kind: "message";
81
+ readonly T: () => MessageReference;
82
+ };
83
+ export interface OneofFieldDesc {
84
+ readonly kind: "oneof";
85
+ readonly name: string;
86
+ readonly variants: readonly VariantDesc[];
87
+ }
88
+ export type FieldDesc = ScalarFieldDesc | MessageFieldDesc | EnumFieldDesc | MapFieldDesc | OneofFieldDesc;
89
+ /** Creates a high-performance, lazy protobuf message codec from an IR field descriptor list. */
90
+ export declare function pb<T extends ProtoMessage = ProtoMessage>(typeName: string, fields?: readonly FieldDesc[]): MessageCodec<T>;
91
+ /** Creates a message using its codec's protobuf defaults. */
92
+ export declare function create<TMessage extends ProtoMessage>(codec: MessageCodec<TMessage>, value?: Partial<TMessage>): TMessage;
93
+ /** Encodes a message using its codec. */
94
+ export declare function toBinary<TMessage extends ProtoMessage>(codec: MessageCodec<TMessage>, value: TMessage): Uint8Array;
95
+ /** Decodes wire bytes using a message codec. */
96
+ export declare function fromBinary<TMessage extends ProtoMessage>(codec: MessageCodec<TMessage>, value: Uint8Array): TMessage;
97
+ /** Converts a message to protobuf JSON using its codec. */
98
+ export declare function toJson<TMessage extends ProtoMessage>(codec: MessageCodec<TMessage>, value: TMessage): JsonValue;
99
+ /** Encodes a JSON value as `google.protobuf.Value`. */
100
+ export declare function encodeJsonValue(value: JsonValue): Uint8Array;
101
+ /** Decodes `google.protobuf.Value` wire bytes into a JSON value. */
102
+ export declare function decodeJsonValue(value: Uint8Array): JsonValue;
@@ -30,6 +30,17 @@ export declare const isClaudeModelId: (modelId: string) => boolean;
30
30
  export declare const isAnthropicNamespacedModelId: (modelId: string) => boolean;
31
31
  /** Qwen family ids (substring match — Qwen SKUs have no stable prefix shape). */
32
32
  export declare const isQwenModelId: (modelId: string) => boolean;
33
+ /**
34
+ * Open-weight Qwen 3.8+ releases (`qwen3.8-27b`, `qwen3.8-2.4t-a95b`, GGUF
35
+ * names like `Qwen3.8-27B-UD-Q6_K_XL`) whose chat template steers thinking
36
+ * depth through a `reasoning_effort` template kwarg (`low`/`medium`/`xhigh`,
37
+ * template default `xhigh`; thinking itself cannot be disabled). Compared
38
+ * component-wise so `qwen3.10` sorts after `qwen3.8`. API-only `-max` SKUs are
39
+ * excluded — Dashscope drives them through OpenAI-style `reasoning_effort`
40
+ * with curated compat. The trailing guard rejects parameter-count lookalikes
41
+ * (`qwen-3.8b`) without breaking `qwen3.8-27b`.
42
+ */
43
+ export declare const isQwen38PlusTemplateEffortModelId: (modelId: string) => boolean;
33
44
  /** Gemma open-weights family (`gemma-3-27b-it`, `google/gemma-4-E2B-it`, `gemma2-9b`). */
34
45
  export declare const isGemmaModelId: (modelId: string) => boolean;
35
46
  /** DeepSeek family by id or display name (proxies often rename the id but keep the name). */
@@ -45,6 +56,8 @@ export declare const isDeepseekModelIdOrName: (modelId: string) => boolean;
45
56
  export declare const isDeepseekV4FlashModelId: (modelId: string) => boolean;
46
57
  /** Xiaomi MiMo family by id or display name. */
47
58
  export declare const isMimoModelIdOrName: (modelId: string) => boolean;
59
+ /** StepFun Step 3.7 Flash SKU in any namespace form (`kilo/stepfun/step-3.7-flash:free`). */
60
+ export declare const isStep37FlashModelId: (modelId: string) => boolean;
48
61
  /** Gemini family ids in any namespace form (`gemini-*`, `google/gemini-*`, `openrouter/google/gemini-…`). */
49
62
  export declare const isGeminiModelId: (modelId: string) => boolean;
50
63
  /** Grok family ids across namespace and delimiter forms (`grok-*`, `cursor-grok-*`, `xai/grok-*`). */
@@ -52,7 +65,8 @@ export declare const isGrokModelId: (modelId: string) => boolean;
52
65
  /**
53
66
  * Grok SKUs that expose the wire `reasoning.effort` dial. Other Grok reasoners
54
67
  * (e.g. `grok-build`, `grok-4.20-0309-reasoning`) think natively but reject the
55
- * param, so callers must omit reasoning effort for them.
68
+ * param, so callers must omit reasoning effort for them. `grok-4.6` accepts
69
+ * `low`/`medium`/`high`/`xhigh` and 400s on `max`.
56
70
  */
57
71
  export declare const isGrokReasoningEffortCapable: (modelId: string) => boolean;
58
72
  /**
@@ -187,6 +201,14 @@ export declare const hasOpus47ApiRestrictions: (modelId: string) => boolean;
187
201
  * @see https://platform.claude.com/docs/en/build-with-claude/mid-conversation-system-messages
188
202
  */
189
203
  export declare const supportsMidConversationSystemMessages: (modelId: string) => boolean;
204
+ /**
205
+ * Models that reliably follow the hashline line-anchored edit dialect
206
+ * (`[path#TAG]` headers plus 1-indexed anchors). Kimi, MiMo, DeepSeek V4
207
+ * Flash, and Step 3.7 Flash miscount anchors or drop the tag header often
208
+ * enough that hosts fall back to a literal search-replace edit format for
209
+ * them.
210
+ */
211
+ export declare const supportsHashlineEdits: (modelId: string) => boolean;
190
212
  export declare const isAnthropicFableOrMythosModel: (modelId: string) => boolean;
191
213
  /** Thinking-variant token location inside a model id. */
192
214
  export interface ThinkingVariantToken {
@@ -6,6 +6,7 @@ export * from "./identity/index.js";
6
6
  export * from "./model-cache.js";
7
7
  export * from "./model-manager.js";
8
8
  export * from "./model-thinking.js";
9
+ export * from "./model-tokenizer.js";
9
10
  export * from "./models.js";
10
11
  export * from "./provider-models/index.js";
11
12
  export * from "./types.js";
@@ -44,8 +44,12 @@ export declare function getSupportedEfforts<TApi extends Api>(model: ApiModel<TA
44
44
  */
45
45
  export declare function clampThinkingLevelForModel<TApi extends Api>(model: ApiModel<TApi> | undefined, requested: Effort | undefined): Effort | undefined;
46
46
  export declare function requireSupportedEffort<TApi extends Api>(model: ApiModel<TApi>, effort: Effort): Effort;
47
- /** Maps a normalized thinking effort to Google's `thinkingLevel` enum values. */
48
- export declare function mapEffortToGoogleThinkingLevel(effort: Effort): "MINIMAL" | "LOW" | "MEDIUM" | "HIGH";
47
+ /** Maps a normalized thinking effort to Google's `thinkingLevel` enum values.
48
+ * When a collapsed family routes `minimal` onto the same wire id as `low`
49
+ * (Antigravity Gemini 3.6/3.7 Flash), emit `LOW` — Cloud Code Assist rejects
50
+ * `MINIMAL` on those `-low` SKUs.
51
+ */
52
+ export declare function mapEffortToGoogleThinkingLevel<TApi extends Api>(effort: Effort, model?: ApiModel<TApi>): "MINIMAL" | "LOW" | "MEDIUM" | "HIGH";
49
53
  /**
50
54
  * Maps a normalized thinking effort to Anthropic adaptive effort values via
51
55
  * the model's baked `thinking.effortMap` (identity for unmapped efforts).
@@ -0,0 +1,8 @@
1
+ import type { ModelTokenizer } from "./types.js";
2
+ /**
3
+ * Resolve the exact locally embedded tokenizer for a canonical model id.
4
+ *
5
+ * This is catalog policy, not a runtime caller heuristic: [`buildModel`](./build.ts)
6
+ * materializes the result as `Model.tokenizer`; consumers read that property.
7
+ */
8
+ export declare function resolveModelTokenizer(modelId: string): ModelTokenizer | undefined;
@@ -428,6 +428,7 @@ export declare const CATALOG_PROVIDERS: readonly [{
428
428
  readonly defaultModel: "openai/gpt-oss-120b";
429
429
  readonly envVars: readonly ["COREWEAVE_API_KEY", "WANDB_API_KEY"];
430
430
  readonly createModelManagerOptions: (config: ModelManagerConfig) => import("../index.js").ModelManagerOptions<"openai-completions", unknown>;
431
+ readonly dynamicModelsAuthoritative: true;
431
432
  readonly catalogDiscovery: {
432
433
  readonly label: "CoreWeave Serverless Inference";
433
434
  };
@@ -220,6 +220,16 @@ export interface OpenAICompat {
220
220
  * Non-Qwen templates ignore the flag, so the auto-detection is safe.
221
221
  */
222
222
  qwenPreserveThinking?: boolean;
223
+ /**
224
+ * Route the requested thinking effort onto the Qwen 3.8+ chat template's
225
+ * `reasoning_effort` kwarg (`low`/`medium`/`xhigh`; template default
226
+ * `xhigh`). Emitted inside `chat_template_kwargs` for both Qwen dialects
227
+ * (plus the top-level field on the `qwen` dialect, which newer llama.cpp
228
+ * builds map natively). Without it the qwen dialects only toggle
229
+ * `enable_thinking` and the template always thinks at its `xhigh` default.
230
+ * Default: auto-detected (Qwen 3.8+ id on a local llama.cpp-style backend).
231
+ */
232
+ qwenTemplateReasoningEffort?: boolean;
223
233
  /** Whether assistant tool-call messages must include non-empty content. Default: false. */
224
234
  requiresAssistantContentForToolCalls?: boolean;
225
235
  /** Whether the provider supports the `tool_choice` parameter. Default: true. */
@@ -558,6 +568,7 @@ export interface ResolvedOpenAISharedCompat {
558
568
  allowsSyntheticReasoningContentForToolCalls: boolean;
559
569
  replayReasoningContent: boolean;
560
570
  qwenPreserveThinking: boolean;
571
+ qwenTemplateReasoningEffort: boolean;
561
572
  requiresThinkingAsText: boolean;
562
573
  requiresMistralToolIds: boolean;
563
574
  requiresToolResultName: boolean;
@@ -595,7 +606,7 @@ export interface ResolvedOpenAISharedCompat {
595
606
  * `buildModel`; request handlers read fields and never detect, resolve, or
596
607
  * allocate.
597
608
  */
598
- export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat & Required<Omit<OpenAICompat, "supportsDeveloperRole" | "supportsReasoningEffort" | "reasoningEffortMap" | "supportsReasoningParams" | "supportsSamplingParams" | "supportsPenaltyAndStopParams" | "thinkingFormat" | "kimiApiFormat" | "reasoningDisableMode" | "omitReasoningEffort" | "includeEncryptedReasoning" | "filterReasoningHistory" | "disableReasoningOnForcedToolChoice" | "disableReasoningOnToolChoice" | "supportsToolChoice" | "supportsForcedToolChoice" | "supportsNamedToolChoice" | "reasoningContentField" | "requiresReasoningContentForToolCalls" | "requiresReasoningContentForAllAssistantTurns" | "allowsSyntheticReasoningContentForToolCalls" | "replayReasoningContent" | "qwenPreserveThinking" | "requiresThinkingAsText" | "requiresMistralToolIds" | "requiresToolResultName" | "requiresAssistantAfterToolResult" | "requiresAssistantContentForToolCalls" | "stripDeepseekSpecialTokens" | "streamMarkupHealingPattern" | "reasoningDeltasMayBeCumulative" | "emptyLengthFinishIsContextError" | "usesOpenAIToolCallIdLimit" | "promptCacheSessionHeader" | "supportsPromptCacheBreakpoints" | "promptCacheBreakpointTtl" | "openRouterRouting" | "isOpenRouterHost" | "supportsStrictMode" | "supportsLongPromptCacheRetention" | "alwaysSendMaxTokens" | "wireModelIdMode" | "vercelGatewayRouting" | "extraBody" | "toolStrictMode" | "toolSchemaFlavor" | "streamFirstEventTimeoutMs" | "streamIdleTimeoutMs" | "cacheControlFormat" | "thinkingKeep" | "strictResponsesPairing" | "supportsImageDetailOriginal" | "whenThinking">> & {
609
+ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat & Required<Omit<OpenAICompat, "supportsDeveloperRole" | "supportsReasoningEffort" | "reasoningEffortMap" | "supportsReasoningParams" | "supportsSamplingParams" | "supportsPenaltyAndStopParams" | "thinkingFormat" | "kimiApiFormat" | "reasoningDisableMode" | "omitReasoningEffort" | "includeEncryptedReasoning" | "filterReasoningHistory" | "disableReasoningOnForcedToolChoice" | "disableReasoningOnToolChoice" | "supportsToolChoice" | "supportsForcedToolChoice" | "supportsNamedToolChoice" | "reasoningContentField" | "requiresReasoningContentForToolCalls" | "requiresReasoningContentForAllAssistantTurns" | "allowsSyntheticReasoningContentForToolCalls" | "replayReasoningContent" | "qwenPreserveThinking" | "qwenTemplateReasoningEffort" | "requiresThinkingAsText" | "requiresMistralToolIds" | "requiresToolResultName" | "requiresAssistantAfterToolResult" | "requiresAssistantContentForToolCalls" | "stripDeepseekSpecialTokens" | "streamMarkupHealingPattern" | "reasoningDeltasMayBeCumulative" | "emptyLengthFinishIsContextError" | "usesOpenAIToolCallIdLimit" | "promptCacheSessionHeader" | "supportsPromptCacheBreakpoints" | "promptCacheBreakpointTtl" | "openRouterRouting" | "isOpenRouterHost" | "supportsStrictMode" | "supportsLongPromptCacheRetention" | "alwaysSendMaxTokens" | "wireModelIdMode" | "vercelGatewayRouting" | "extraBody" | "toolStrictMode" | "toolSchemaFlavor" | "streamFirstEventTimeoutMs" | "streamIdleTimeoutMs" | "cacheControlFormat" | "thinkingKeep" | "strictResponsesPairing" | "supportsImageDetailOriginal" | "whenThinking">> & {
599
610
  vercelGatewayRouting?: OpenAICompat["vercelGatewayRouting"];
600
611
  extraBody?: OpenAICompat["extraBody"];
601
612
  cacheControlFormat?: OpenAICompat["cacheControlFormat"];
@@ -705,6 +716,15 @@ export interface LongContextTokenCost extends TokenCost {
705
716
  export interface ModelCost extends TokenCost {
706
717
  longContext?: LongContextTokenCost;
707
718
  }
719
+ /**
720
+ * Exact local content tokenizer family for a model.
721
+ *
722
+ * Absent means no first-party local tokenizer is known and consumers retain
723
+ * their estimate/default-tokenizer policy. The values name tokenizer
724
+ * generations rather than providers: DeepSeek V3 through V4 share
725
+ * `"deepseek-v3"`; Kimi K2 through K3 share `"kimi-k2"`.
726
+ */
727
+ export type ModelTokenizer = "claude-v3" | "claude-v47" | "claude-v5" | "claude-v5-sonnet" | "qwen3" | "deepseek-v3" | "kimi-k2" | "glm5";
708
728
  export interface Model<TApi extends Api = Api> {
709
729
  id: string;
710
730
  /**
@@ -728,6 +748,12 @@ export interface Model<TApi extends Api = Api> {
728
748
  provider: Provider;
729
749
  baseUrl: string;
730
750
  reasoning: boolean;
751
+ /**
752
+ * Exact local tokenizer family resolved from the model identity or supplied
753
+ * explicitly by a catalog/discovery source. Absent leaves local counting to
754
+ * the consumer's fallback policy.
755
+ */
756
+ tokenizer?: ModelTokenizer;
731
757
  input: ("text" | "image")[];
732
758
  /**
733
759
  * Decoder family used for image inputs when it has narrower format support
@@ -58,6 +58,8 @@ export declare const ANTIGRAVITY_VARIANT_COLLAPSE_TABLE: VariantCollapseTable;
58
58
  /** `google-gemini-cli` Gemini families on the official CLI's level transport. */
59
59
  export declare const GEMINI_CLI_VARIANT_COLLAPSE_TABLE: VariantCollapseTable;
60
60
  export declare const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable;
61
+ /** `cursor` Grok families: per-effort siblings collapsed per service-tier lane. */
62
+ export declare const CURSOR_VARIANT_COLLAPSE_TABLE: VariantCollapseTable;
61
63
  /** Provider id → hand collapse table. The CCA providers diverge on thinking transport. */
62
64
  export declare const VARIANT_COLLAPSE_TABLES: Readonly<Record<string, VariantCollapseTable>>;
63
65
  /**
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@oh-my-pi/pi-catalog",
4
- "version": "17.3.7",
4
+ "version": "17.4.0",
5
5
  "description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence",
6
6
  "homepage": "https://omp.sh",
7
7
  "author": "Can Boluk",
@@ -31,15 +31,14 @@
31
31
  "fix": "biome check --write --unsafe .",
32
32
  "fmt": "biome format --write .",
33
33
  "gen:models": "bun scripts/generate-models.ts",
34
- "gen:cursor-proto": "protoc --plugin=protoc-gen-es=../../node_modules/.bin/protoc-gen-es --es_out=src/discovery/cursor-gen --es_opt=target=ts -I ../ai/src/providers/cursor/proto ../ai/src/providers/cursor/proto/agent.proto"
34
+ "gen:proto": "bun scripts/generate-protocols.ts"
35
35
  },
36
36
  "dependencies": {
37
- "@bufbuild/protobuf": "^2.12.1",
38
- "@oh-my-pi/omptype": "17.3.7",
39
- "@oh-my-pi/pi-utils": "17.3.7"
37
+ "@oh-my-pi/omptype": "17.4.0",
38
+ "@oh-my-pi/pi-utils": "17.4.0"
40
39
  },
41
40
  "devDependencies": {
42
- "@oh-my-pi/pi-ai": "17.3.7",
41
+ "@oh-my-pi/pi-ai": "17.4.0",
43
42
  "@types/bun": "^1.3.14"
44
43
  },
45
44
  "engines": {
package/src/build.ts CHANGED
@@ -16,6 +16,7 @@ import { buildDevinCompat } from "./compat/devin";
16
16
  import { buildOpenAICompat, buildOpenAIResponsesCompat, buildOpenRouterCompat } from "./compat/openai";
17
17
  import { bareModelId, parseOpenAIModel, semverGte } from "./identity/classify";
18
18
  import { resolveModelThinking } from "./model-thinking";
19
+ import { resolveModelTokenizer } from "./model-tokenizer";
19
20
  import type { Api, CompatOf, Model, ModelSpec } from "./types";
20
21
  import { cleanModelName } from "./utils";
21
22
 
@@ -58,12 +59,18 @@ function supportsOpenAIGAComputerUse(spec: ModelSpec<Api>, explicitSupport: bool
58
59
  return parsed !== null && semverGte(parsed.version, "5.4");
59
60
  }
60
61
 
62
+ /**
63
+ * Build one model from an authored spec. Bundled models.json rows are fully
64
+ * materialized by the generator and consumed directly (see `models.ts`), so
65
+ * this only runs for discovered/custom/override specs.
66
+ */
61
67
  export function buildModel<TApi extends Api>(spec: ModelSpec<TApi>): Model<TApi> {
62
68
  const compat = buildCompat(spec) as CompatOf<TApi>;
63
69
  const supportsComputerUseConfig = explicitComputerUseConfig(spec);
64
70
  return {
65
71
  ...spec,
66
72
  name: cleanModelName(spec.name),
73
+ tokenizer: spec.tokenizer ?? resolveModelTokenizer(spec.requestModelId ?? spec.id),
67
74
  thinking: resolveModelThinking(spec, compat),
68
75
  supportsComputerUse: supportsOpenAIGAComputerUse(spec, supportsComputerUseConfig),
69
76
  supportsComputerUseConfig,
@@ -22,6 +22,7 @@ import {
22
22
  isKimiModelId,
23
23
  isMimoModelIdOrName,
24
24
  isOpenAISamplingRestrictedModelId,
25
+ isQwen38PlusTemplateEffortModelId,
25
26
  isQwenModelId,
26
27
  } from "../identity/family";
27
28
  import type {
@@ -464,7 +465,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
464
465
  ? "zai"
465
466
  : isOpenRouter
466
467
  ? "openrouter"
467
- : isQwen && isNvidiaNim
468
+ : isQwen && (isNvidiaNim || provider === "vllm")
468
469
  ? "qwen-chat-template"
469
470
  : isQwen && isFireworks
470
471
  ? "openai"
@@ -587,6 +588,18 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
587
588
  // parameter, so the flag stays a no-op outside the Qwen path.
588
589
  qwenPreserveThinking:
589
590
  (thinkingFormat === "qwen" || thinkingFormat === "qwen-chat-template") && isLocalOpenAICompatBackend,
591
+ // Qwen 3.8+ templates steer thinking depth via the `reasoning_effort`
592
+ // template kwarg (low/medium/xhigh, default xhigh); without routing the
593
+ // requested effort there, the enable_thinking toggle alone leaves the
594
+ // model at xhigh no matter what the user selects.
595
+ // Local-only like `qwenPreserveThinking`: first-party Qwen APIs
596
+ // (Dashscope, Qwen Portal) drive effort through their own OpenAI-style
597
+ // dialect, and local Ollama keeps its native effort vocabulary.
598
+ qwenTemplateReasoningEffort:
599
+ (thinkingFormat === "qwen" || thinkingFormat === "qwen-chat-template") &&
600
+ isLocalOpenAICompatBackend &&
601
+ provider !== "ollama" &&
602
+ isQwen38PlusTemplateEffortModelId(spec.id),
590
603
  requiresAssistantContentForToolCalls: isKimiModel || isDirectDeepseekReasoning,
591
604
  cacheControlFormat: isOpenRouter && spec.id.startsWith("anthropic/") ? "anthropic" : undefined,
592
605
  supportsPromptCacheBreakpoints,
@@ -761,6 +774,7 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
761
774
  // Responses-only; the Qwen `preserve_thinking` template knob lives on
762
775
  // the chat-completions wire shape, never on Responses.
763
776
  qwenPreserveThinking: false,
777
+ qwenTemplateReasoningEffort: false,
764
778
  requiresThinkingAsText: false,
765
779
  requiresMistralToolIds: false,
766
780
  requiresToolResultName: false,
@@ -807,6 +821,18 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
807
821
  if (spec.compat?.omitReasoningEffort === undefined && !compat.supportsReasoningEffort) {
808
822
  compat.omitReasoningEffort = true;
809
823
  }
824
+ // xai-oauth cache/discovery rows written before a SKU joined the
825
+ // effort-capable allowlist still carry omitReasoningEffort: true. The
826
+ // allowlist is the live wire contract; do not let that stale flag hide
827
+ // the picker or strip reasoning.effort.
828
+ if (
829
+ spec.provider === "xai-oauth" &&
830
+ isGrokReasoningEffortCapable(id) &&
831
+ spec.compat?.supportsReasoningEffort !== false
832
+ ) {
833
+ compat.supportsReasoningEffort = true;
834
+ compat.omitReasoningEffort = false;
835
+ }
810
836
  return compat;
811
837
  }
812
838
 
@@ -1,5 +1,6 @@
1
1
  import { type } from "@oh-my-pi/omptype";
2
2
  import { parseKnownModel, semverEqual } from "../identity/classify";
3
+ import { getBundledModels } from "../models";
3
4
  import { resolveOpenAIDaybreakStandardCost } from "../openai-pricing";
4
5
  import type { FetchImpl, ModelSpec } from "../types";
5
6
  import { discoveryFetch } from "../utils";
@@ -22,6 +23,29 @@ const GPT_5_6_CONTEXT_WINDOW = 372_000;
22
23
  */
23
24
  const GPT_5_6_1M_CONTEXT_WINDOW = 1_000_000;
24
25
  const CODEX_GPT_5_6_1M_SLUGS: ReadonlySet<string> = new Set(["gpt-5.6-luna", "gpt-5.6-sol", "gpt-5.6-terra"]);
26
+ /**
27
+ * Codex advertises worker-mode SKUs under a `-wm` suffix (`gpt-5.6-luna-wm`).
28
+ *
29
+ * Those rows route through the same Codex backend as their plain SKU, but an
30
+ * authoritative discovery list that only advertises the `-wm` slug prunes the
31
+ * bundled plain model, leaving a configured `openai-codex/gpt-5.6-luna`
32
+ * unresolvable except via fuzzy fallback onto the `-wm` row — which this user's
33
+ * ChatGPT account rejects. The compatibility rule, scoped to Codex discovery:
34
+ * a `-wm` slug whose plain counterpart exists in the bundled Codex catalog is
35
+ * ALSO registered under its plain id. Both listings derive their base-model
36
+ * metadata (1M-window floor, daybreak pricing, context fallback) from the
37
+ * canonical plain slug — the suffix is a routing variant, not a different
38
+ * model, so the `-wm` row no longer keeps stale backend-parsed capability
39
+ * values while its plain listing is enriched.
40
+ *
41
+ * Deliberate boundary: the "safe" gate is the bundled Codex catalog. A `-wm`
42
+ * slug whose plain counterpart is only a user-local models.yml entry (not
43
+ * bundled) stays verbatim — authoritative discovery for genuinely distinct
44
+ * `-wm` SKUs is preserved, and a hidden plain backend entry can be re-surfaced
45
+ * through its advertised `-wm` row because the configured plain slug must
46
+ * resolve.
47
+ */
48
+ const CODEX_WORKER_SUFFIX = "-wm";
25
49
  const CODEX_REMOTE_COMPACTION = {
26
50
  enabled: true,
27
51
  api: "openai-codex-responses",
@@ -194,11 +218,30 @@ function normalizeCodexModels(payload: unknown, baseUrl: string): ModelSpec<"ope
194
218
  }
195
219
 
196
220
  const entries = parsedResponse.models ?? parsedResponse.data ?? [];
197
- const normalized: NormalizedCodexModel[] = [];
221
+ const parsedEntries: ParsedCodexModelEntry[] = [];
198
222
  for (const entry of entries) {
199
- const model = normalizeCodexModelEntry(entry, baseUrl);
200
- if (model) {
201
- normalized.push(model);
223
+ const parsed = parseCodexModelEntry(entry);
224
+ if (parsed) {
225
+ parsedEntries.push(parsed);
226
+ }
227
+ }
228
+
229
+ // A worker `-wm` slug gets an extra plain-id route only when the bundled
230
+ // catalog ships the plain SKU (the "safe" precondition); the backend's own
231
+ // plain slug wins over any synthesized clone, and unknown `-wm` SKUs stay
232
+ // verbatim. Both listings of a safe `-wm` model carry the same base-model
233
+ // metadata (context-window floor, daybreak pricing) derived from the
234
+ // canonical plain slug — the suffix is a routing variant, not a different
235
+ // model.
236
+ const advertisedSlugs = new Set(parsedEntries.map(parsed => parsed.slug));
237
+ const bundledCodexModelIds = getBundledCodexModelIds();
238
+ const normalized: NormalizedCodexModel[] = [];
239
+ for (const parsed of parsedEntries) {
240
+ const canonicalSlug = plainCounterpartForWorkerSlug(parsed.slug, bundledCodexModelIds) ?? parsed.slug;
241
+ normalized.push(buildNormalizedCodexModel(parsed, parsed.slug, canonicalSlug, baseUrl));
242
+ const plainSlug = canonicalSlug !== parsed.slug ? canonicalSlug : null;
243
+ if (plainSlug && !advertisedSlugs.has(plainSlug)) {
244
+ normalized.push(buildNormalizedCodexModel(parsed, plainSlug, canonicalSlug, baseUrl));
202
245
  }
203
246
  }
204
247
 
@@ -212,7 +255,37 @@ function normalizeCodexModels(payload: unknown, baseUrl: string): ModelSpec<"ope
212
255
  return normalized.map(item => item.model);
213
256
  }
214
257
 
215
- function normalizeCodexModelEntry(entry: unknown, baseUrl: string): NormalizedCodexModel | null {
258
+ /** Ids of the bundled Codex catalog, consulted once per discovery run. */
259
+ function getBundledCodexModelIds(): ReadonlySet<string> {
260
+ const ids = new Set(getBundledModels("openai-codex").map(model => model.id));
261
+ return ids;
262
+ }
263
+
264
+ /**
265
+ * Map a Codex worker `-wm` slug to its plain counterpart when the bundled
266
+ * catalog registers that plain SKU. Returns `null` for non-worker slugs and
267
+ * for `-wm` slugs without a safe plain counterpart.
268
+ */
269
+ function plainCounterpartForWorkerSlug(slug: string, bundledCodexModelIds: ReadonlySet<string>): string | null {
270
+ if (!slug.endsWith(CODEX_WORKER_SUFFIX)) {
271
+ return null;
272
+ }
273
+ const plain = slug.slice(0, -CODEX_WORKER_SUFFIX.length);
274
+ return plain.length > 0 && bundledCodexModelIds.has(plain) ? plain : null;
275
+ }
276
+
277
+ interface ParsedCodexModelEntry {
278
+ slug: string;
279
+ name: string;
280
+ contextWindow: number | null;
281
+ reasoning: boolean;
282
+ input: ("text" | "image")[];
283
+ preferWebsockets: boolean;
284
+ useResponsesLite: boolean;
285
+ priority: number;
286
+ }
287
+
288
+ function parseCodexModelEntry(entry: unknown): ParsedCodexModelEntry | null {
216
289
  const parsedEntry = codexModelEntrySchema(entry);
217
290
  if (parsedEntry instanceof type.errors) {
218
291
  return null;
@@ -229,44 +302,65 @@ function normalizeCodexModelEntry(entry: unknown, baseUrl: string): NormalizedCo
229
302
  return null;
230
303
  }
231
304
 
232
- const name = toNonEmptyString(payload.display_name) ?? slug;
305
+ return {
306
+ slug,
307
+ name: toNonEmptyString(payload.display_name) ?? slug,
308
+ contextWindow: toPositiveInt(payload.context_window),
309
+ reasoning: supportsReasoning(payload.default_reasoning_level, payload.supported_reasoning_levels),
310
+ input: normalizeInputModalities(payload.input_modalities),
311
+ preferWebsockets: toBoolean(payload.prefer_websockets) === true,
312
+ useResponsesLite: toBoolean(payload.use_responses_lite) === true,
313
+ priority: toFiniteNumber(payload.priority) ?? Number.MAX_SAFE_INTEGER,
314
+ };
315
+ }
316
+
317
+ /**
318
+ * Build a normalized Codex model spec. `slug` is the registered id (either the
319
+ * advertised slug or a synthesized plain counterpart); `canonicalSlug` names
320
+ * the model's bundled SKU (`slug` itself for plain/unknown rows, the plain
321
+ * counterpart for a safe `-wm` row) and owns the base-model metadata derivation
322
+ * so both listings of a model report the same context window and pricing.
323
+ */
324
+ function buildNormalizedCodexModel(
325
+ parsed: ParsedCodexModelEntry,
326
+ slug: string,
327
+ canonicalSlug: string,
328
+ baseUrl: string,
329
+ ): NormalizedCodexModel {
233
330
  // Codex discovery historically omitted `context_window` for GPT-5.6-family
234
331
  // SKUs (#5705); luna/sol/terra additionally floor the reported value because
235
- // the registry still declares the pre-1M 272000 window.
236
- const parsed = parseKnownModel(slug);
332
+ // the registry still declares the pre-1M 272000 window. Keyed on the
333
+ // canonical slug so a safe `gpt-5.6-luna-wm` row gets the same floor as its
334
+ // plain listing.
335
+ const parsedKnown = parseKnownModel(canonicalSlug);
237
336
  const fallbackContextWindow =
238
- parsed.family === "openai" && semverEqual(parsed.version, "5.6")
337
+ parsedKnown.family === "openai" && semverEqual(parsedKnown.version, "5.6")
239
338
  ? GPT_5_6_CONTEXT_WINDOW
240
339
  : DEFAULT_CONTEXT_WINDOW;
241
- const reportedContextWindow = toPositiveInt(payload.context_window) ?? fallbackContextWindow;
242
- const contextWindow = CODEX_GPT_5_6_1M_SLUGS.has(slug)
340
+ const reportedContextWindow = parsed.contextWindow ?? fallbackContextWindow;
341
+ const contextWindow = CODEX_GPT_5_6_1M_SLUGS.has(canonicalSlug)
243
342
  ? Math.max(reportedContextWindow, GPT_5_6_1M_CONTEXT_WINDOW)
244
343
  : reportedContextWindow;
245
344
  const maxTokens = Math.min(DEFAULT_MAX_TOKENS, contextWindow);
246
- const reasoning = supportsReasoning(payload.default_reasoning_level, payload.supported_reasoning_levels);
247
- const input = normalizeInputModalities(payload.input_modalities);
248
- const preferWebsockets = toBoolean(payload.prefer_websockets) === true;
249
- const useResponsesLite = toBoolean(payload.use_responses_lite) === true;
250
- const priority = toFiniteNumber(payload.priority) ?? Number.MAX_SAFE_INTEGER;
251
- const daybreakCost = resolveOpenAIDaybreakStandardCost(slug);
345
+ const daybreakCost = resolveOpenAIDaybreakStandardCost(canonicalSlug);
252
346
 
253
347
  return {
254
- priority,
348
+ priority: parsed.priority,
255
349
  model: {
256
350
  id: slug,
257
- name,
351
+ name: parsed.name,
258
352
  api: "openai-codex-responses",
259
353
  provider: "openai-codex",
260
354
  baseUrl,
261
- reasoning,
262
- input,
355
+ reasoning: parsed.reasoning,
356
+ input: parsed.input,
263
357
  cost: daybreakCost ? { ...daybreakCost } : { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
264
358
  remoteCompaction: CODEX_REMOTE_COMPACTION,
265
359
  contextWindow,
266
360
  maxTokens,
267
- ...(preferWebsockets ? { preferWebsockets: true } : {}),
268
- ...(useResponsesLite ? { useResponsesLite: true } : {}),
269
- ...(priority !== Number.MAX_SAFE_INTEGER ? { priority } : {}),
361
+ ...(parsed.preferWebsockets ? { preferWebsockets: true } : {}),
362
+ ...(parsed.useResponsesLite ? { useResponsesLite: true } : {}),
363
+ ...(parsed.priority !== Number.MAX_SAFE_INTEGER ? { priority: parsed.priority } : {}),
270
364
  },
271
365
  };
272
366
  }