pi-wafer-provider 1.1.3 → 1.1.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -76,14 +76,14 @@ pi
76
76
 
77
77
  | Model | Type | Context | Max Output | Input Cost | Output Cost | Cached Input |
78
78
  |-------|------|---------|------------|------------|-------------|--------------|
79
- | GLM 5.1 | Text | 203K | 33K | $1.50 | $4.50 | $0.15 |
80
- | GLM 5.2 | Text | 1M | 16K | Free | Free | Free |
81
- | Glm5.2 Fast | Text | 1M | 16K | Free | Free | Free |
82
- | Kimi K2.6 | Text | 262K | 33K | $1.10 | $4.80 | $0.11 |
83
- | Kimi K3 | Text | 912K | 16K | Free | Free | Free |
84
- | Kimi K3 Fast | Text | 1M | 16K | Free | Free | Free |
85
- | MiniMax M3 | Text | 1M | 16K | Free | Free | Free |
86
- | Qwen 3.5 397B (A17B) | Text + Image | 262K | 33K | $0.60 | $3.60 | $0.06 |
79
+ | GLM-5.1 | Text | 203K | 33K | $1.00 | $3.20 | $0.10 |
80
+ | GLM-5.2 | Text | 1M | 16K | $1.26 | $3.96 | $0.23 |
81
+ | GLM5.2-Fast | Text | 1M | 16K | $2.10 | $6.60 | $0.21 |
82
+ | Kimi-K2.6 | Text + Image | 262K | 33K | $1.14 | $4.80 | $0.19 |
83
+ | Kimi-K3 | Text + Image | 912K | 16K | $3.00 | $15.00 | $0.30 |
84
+ | Kimi-K3-Fast | Text + Image | 1M | 16K | $4.50 | $22.50 | $0.45 |
85
+ | MiniMax-M3 | Text + Image | 1M | 16K | $0.33 | $1.32 | $0.07 |
86
+ | Qwen 3.5 397B (A17B) | Text | 262K | 33K | Free | Free | Free |
87
87
 
88
88
  *Costs are per million tokens. Prices based on official provider pricing.*
89
89
 
package/index.ts CHANGED
@@ -189,17 +189,29 @@ interface ProviderConfig {
189
189
 
190
190
  /** Transform a model from the Wafer /v1/models API. */
191
191
  function transformApiModel(apiModel: any): JsonModel | null {
192
+ const details = apiModel.wafer || {};
193
+ const capabilities = details.capabilities || {};
194
+ const pricing = details.pricing || {};
195
+ const reasoning = capabilities.reasoning === true;
192
196
  return {
193
197
  id: apiModel.id,
194
- name: apiModel.id,
195
- reasoning: false,
196
- input: ["text"],
197
- cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
198
- contextWindow: apiModel.max_model_len || 0,
199
- maxTokens: 0,
198
+ name: details.display_name || apiModel.id,
199
+ reasoning,
200
+ input: capabilities.vision === true ? ["text", "image"] : ["text"],
201
+ cost: {
202
+ input: (pricing.input_cents_per_million || 0) / 100,
203
+ output: (pricing.output_cents_per_million || 0) / 100,
204
+ cacheRead: (pricing.cache_read_cents_per_million || 0) / 100,
205
+ cacheWrite: 0,
206
+ },
207
+ contextWindow: details.context_length || apiModel.max_model_len || 0,
208
+ maxTokens: details.context_length || apiModel.max_model_len || 0,
200
209
  compat: {
210
+ supportsStore: false,
211
+ supportsDeveloperRole: false,
212
+ maxTokensField: "max_completion_tokens",
201
213
  supportsZdr: apiModel.zdr_supported ?? undefined,
202
- supportsReasoningEffort: true,
214
+ ...(reasoning ? { supportsReasoningEffort: true } : {}),
203
215
  },
204
216
  };
205
217
  }
package/models.json CHANGED
@@ -1,15 +1,15 @@
1
1
  [
2
2
  {
3
3
  "id": "GLM-5.1",
4
- "name": "GLM 5.1",
5
- "reasoning": false,
4
+ "name": "GLM-5.1",
5
+ "reasoning": true,
6
6
  "input": [
7
7
  "text"
8
8
  ],
9
9
  "cost": {
10
- "input": 1.5,
11
- "output": 4.5,
12
- "cacheRead": 0.15,
10
+ "input": 1,
11
+ "output": 3.2,
12
+ "cacheRead": 0.1,
13
13
  "cacheWrite": 0
14
14
  },
15
15
  "contextWindow": 202752,
@@ -17,20 +17,22 @@
17
17
  "compat": {
18
18
  "maxTokensField": "max_completion_tokens",
19
19
  "supportsDeveloperRole": false,
20
- "supportsZdr": true
20
+ "supportsZdr": true,
21
+ "supportsStore": false,
22
+ "supportsReasoningEffort": true
21
23
  }
22
24
  },
23
25
  {
24
26
  "id": "GLM-5.2",
25
- "name": "GLM 5.2",
26
- "reasoning": false,
27
+ "name": "GLM-5.2",
28
+ "reasoning": true,
27
29
  "input": [
28
30
  "text"
29
31
  ],
30
32
  "cost": {
31
- "input": 0,
32
- "output": 0,
33
- "cacheRead": 0,
33
+ "input": 1.26,
34
+ "output": 3.96,
35
+ "cacheRead": 0.23,
34
36
  "cacheWrite": 0
35
37
  },
36
38
  "contextWindow": 1048576,
@@ -39,20 +41,21 @@
39
41
  "maxTokensField": "max_completion_tokens",
40
42
  "supportsDeveloperRole": false,
41
43
  "supportsStore": false,
42
- "supportsZdr": true
44
+ "supportsZdr": true,
45
+ "supportsReasoningEffort": true
43
46
  }
44
47
  },
45
48
  {
46
49
  "id": "glm5.2-fast",
47
- "name": "Glm5.2 Fast",
48
- "reasoning": false,
50
+ "name": "GLM5.2-Fast",
51
+ "reasoning": true,
49
52
  "input": [
50
53
  "text"
51
54
  ],
52
55
  "cost": {
53
- "input": 0,
54
- "output": 0,
55
- "cacheRead": 0,
56
+ "input": 2.1,
57
+ "output": 6.6,
58
+ "cacheRead": 0.21,
56
59
  "cacheWrite": 0
57
60
  },
58
61
  "contextWindow": 1048576,
@@ -61,20 +64,22 @@
61
64
  "maxTokensField": "max_completion_tokens",
62
65
  "supportsDeveloperRole": false,
63
66
  "supportsStore": false,
64
- "supportsZdr": true
67
+ "supportsZdr": true,
68
+ "supportsReasoningEffort": true
65
69
  }
66
70
  },
67
71
  {
68
72
  "id": "Kimi-K2.6",
69
- "name": "Kimi K2.6",
70
- "reasoning": false,
73
+ "name": "Kimi-K2.6",
74
+ "reasoning": true,
71
75
  "input": [
72
- "text"
76
+ "text",
77
+ "image"
73
78
  ],
74
79
  "cost": {
75
- "input": 1.1,
80
+ "input": 1.14,
76
81
  "output": 4.8,
77
- "cacheRead": 0.11,
82
+ "cacheRead": 0.19,
78
83
  "cacheWrite": 0
79
84
  },
80
85
  "contextWindow": 262144,
@@ -82,20 +87,23 @@
82
87
  "compat": {
83
88
  "maxTokensField": "max_completion_tokens",
84
89
  "supportsDeveloperRole": false,
85
- "supportsZdr": false
90
+ "supportsZdr": false,
91
+ "supportsStore": false,
92
+ "supportsReasoningEffort": true
86
93
  }
87
94
  },
88
95
  {
89
96
  "id": "Kimi-K3",
90
- "name": "Kimi K3",
91
- "reasoning": false,
97
+ "name": "Kimi-K3",
98
+ "reasoning": true,
92
99
  "input": [
93
- "text"
100
+ "text",
101
+ "image"
94
102
  ],
95
103
  "cost": {
96
- "input": 0,
97
- "output": 0,
98
- "cacheRead": 0,
104
+ "input": 3,
105
+ "output": 15,
106
+ "cacheRead": 0.3,
99
107
  "cacheWrite": 0
100
108
  },
101
109
  "contextWindow": 912384,
@@ -104,20 +112,22 @@
104
112
  "maxTokensField": "max_completion_tokens",
105
113
  "supportsDeveloperRole": false,
106
114
  "supportsStore": false,
107
- "supportsZdr": true
115
+ "supportsZdr": true,
116
+ "supportsReasoningEffort": true
108
117
  }
109
118
  },
110
119
  {
111
120
  "id": "kimi-k3-fast",
112
- "name": "Kimi K3 Fast",
113
- "reasoning": false,
121
+ "name": "Kimi-K3-Fast",
122
+ "reasoning": true,
114
123
  "input": [
115
- "text"
124
+ "text",
125
+ "image"
116
126
  ],
117
127
  "cost": {
118
- "input": 0,
119
- "output": 0,
120
- "cacheRead": 0,
128
+ "input": 4.5,
129
+ "output": 22.5,
130
+ "cacheRead": 0.45,
121
131
  "cacheWrite": 0
122
132
  },
123
133
  "contextWindow": 1048576,
@@ -126,20 +136,22 @@
126
136
  "maxTokensField": "max_completion_tokens",
127
137
  "supportsDeveloperRole": false,
128
138
  "supportsStore": false,
129
- "supportsZdr": true
139
+ "supportsZdr": true,
140
+ "supportsReasoningEffort": true
130
141
  }
131
142
  },
132
143
  {
133
144
  "id": "MiniMax-M3",
134
- "name": "MiniMax M3",
135
- "reasoning": false,
145
+ "name": "MiniMax-M3",
146
+ "reasoning": true,
136
147
  "input": [
137
- "text"
148
+ "text",
149
+ "image"
138
150
  ],
139
151
  "cost": {
140
- "input": 0,
141
- "output": 0,
142
- "cacheRead": 0,
152
+ "input": 0.33,
153
+ "output": 1.32,
154
+ "cacheRead": 0.07,
143
155
  "cacheWrite": 0
144
156
  },
145
157
  "contextWindow": 1048576,
@@ -148,7 +160,8 @@
148
160
  "maxTokensField": "max_completion_tokens",
149
161
  "supportsDeveloperRole": false,
150
162
  "supportsStore": false,
151
- "supportsZdr": false
163
+ "supportsZdr": false,
164
+ "supportsReasoningEffort": true
152
165
  }
153
166
  },
154
167
  {
@@ -156,13 +169,12 @@
156
169
  "name": "Qwen 3.5 397B (A17B)",
157
170
  "reasoning": false,
158
171
  "input": [
159
- "text",
160
- "image"
172
+ "text"
161
173
  ],
162
174
  "cost": {
163
- "input": 0.6,
164
- "output": 3.6,
165
- "cacheRead": 0.06,
175
+ "input": 0,
176
+ "output": 0,
177
+ "cacheRead": 0,
166
178
  "cacheWrite": 0
167
179
  },
168
180
  "contextWindow": 262144,
@@ -170,7 +182,8 @@
170
182
  "compat": {
171
183
  "maxTokensField": "max_completion_tokens",
172
184
  "supportsDeveloperRole": false,
173
- "supportsZdr": false
185
+ "supportsZdr": false,
186
+ "supportsStore": false
174
187
  }
175
188
  }
176
189
  ]
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-wafer-provider",
3
- "version": "1.1.3",
3
+ "version": "1.1.4",
4
4
  "description": "Wafer Serverless provider extension for pi - Access Qwen3.5-397B-A17B, GLM-5.1, Kimi K2.6, and DeepSeek-V4 through the Wafer Serverless API",
5
5
  "type": "module",
6
6
  "main": "index.ts",
package/patch.json CHANGED
@@ -1,6 +1,8 @@
1
1
  {
2
2
  "Qwen3.5-397B-A17B": {
3
- "providers": ["wafer-serverless"],
3
+ "providers": [
4
+ "wafer-serverless"
5
+ ],
4
6
  "reasoning": true,
5
7
  "compat": {
6
8
  "thinkingFormat": "qwen",
@@ -10,7 +12,9 @@
10
12
  }
11
13
  },
12
14
  "GLM-5.1": {
13
- "providers": ["wafer-serverless"],
15
+ "providers": [
16
+ "wafer-serverless"
17
+ ],
14
18
  "reasoning": true,
15
19
  "compat": {
16
20
  "thinkingFormat": "zai",
@@ -20,67 +24,14 @@
20
24
  }
21
25
  },
22
26
  "Kimi-K2.6": {
23
- "providers": ["wafer-serverless"],
27
+ "providers": [
28
+ "wafer-serverless"
29
+ ],
24
30
  "reasoning": true,
25
31
  "compat": {
26
32
  "maxTokensField": "max_completion_tokens",
27
33
  "supportsDeveloperRole": false,
28
34
  "supportsReasoningEffort": true
29
35
  }
30
- },
31
- "Qwen3.6-35B-A3B": {
32
- "providers": ["wafer-serverless"],
33
- "reasoning": true,
34
- "compat": {
35
- "thinkingFormat": "qwen",
36
- "maxTokensField": "max_completion_tokens",
37
- "supportsDeveloperRole": false,
38
- "supportsReasoningEffort": true
39
- }
40
- },
41
- "deepseek-v4-flash": {
42
- "providers": ["wafer-serverless"],
43
- "reasoning": true,
44
- "compat": {
45
- "thinkingFormat": "deepseek",
46
- "maxTokensField": "max_completion_tokens",
47
- "supportsDeveloperRole": false,
48
- "supportsReasoningEffort": true
49
- }
50
- },
51
- "deepseek-v4-pro": {
52
- "providers": ["wafer-serverless"],
53
- "reasoning": true,
54
- "contextWindow": 1000000,
55
- "maxTokens": 384000,
56
- "thinkingLevelMap": {
57
- "minimal": null,
58
- "low": null,
59
- "medium": null,
60
- "high": "high",
61
- "max": "max"
62
- },
63
- "compat": {
64
- "thinkingFormat": "deepseek",
65
- "maxTokensField": "max_completion_tokens",
66
- "supportsDeveloperRole": false,
67
- "supportsReasoningEffort": true
68
- }
69
- },
70
- "qwen3.7-max": {
71
- "providers": ["wafer-serverless"],
72
- "reasoning": true,
73
- "cost": {
74
- "input": 5.0,
75
- "output": 15.0,
76
- "cacheRead": 0.5,
77
- "cacheWrite": 0
78
- },
79
- "compat": {
80
- "thinkingFormat": "qwen",
81
- "maxTokensField": "max_completion_tokens",
82
- "supportsDeveloperRole": false,
83
- "supportsReasoningEffort": true
84
- }
85
36
  }
86
37
  }
@@ -6,11 +6,9 @@
6
6
  * - models.json: Provider model definitions (enriched with pricing & compat)
7
7
  * - README.md: Model table in the Available Models section
8
8
  *
9
- * The Wafer /v1/models API returns basic model info (id, max_model_len)
10
- * but does NOT include pricing or max output tokens.
11
- * models.json is the source of truth for curated specs — the script preserves
12
- * existing data and only adds new models with sensible defaults.
13
- * Curate models.json manually after new model discovery.
9
+ * The Wafer /v1/models API returns model limits plus nested capability, pricing,
10
+ * modality, and ZDR metadata. models.json mirrors those API-owned fields while
11
+ * patch.json remains the source of truth for thinking controls and corrections.
14
12
  *
15
13
  * patch.json is applied at runtime by the provider — not baked into models.json.
16
14
  *
@@ -72,39 +70,51 @@ async function fetchModels() {
72
70
  function transformApiModel(apiModel, existingModelsMap) {
73
71
  const id = apiModel.id;
74
72
 
75
- // Preserve existing curated data (pricing, reasoning, compat, etc.)
73
+ const details = apiModel.wafer || {};
74
+ const capabilities = details.capabilities || {};
75
+ const pricing = details.pricing || {};
76
+ const reasoning = capabilities.reasoning === true;
77
+ const input = capabilities.vision === true ? ['text', 'image'] : ['text'];
78
+ const cost = {
79
+ input: (pricing.input_cents_per_million || 0) / 100,
80
+ output: (pricing.output_cents_per_million || 0) / 100,
81
+ cacheRead: (pricing.cache_read_cents_per_million || 0) / 100,
82
+ cacheWrite: 0,
83
+ };
84
+
76
85
  if (existingModelsMap[id]) {
77
86
  const existing = { ...existingModelsMap[id] };
78
- // Update API-derived fields if changed
79
- if (apiModel.max_model_len) {
80
- existing.contextWindow = apiModel.max_model_len;
81
- }
82
- if (apiModel.zdr_supported !== undefined) {
83
- existing.compat = existing.compat || {};
84
- existing.compat.supportsZdr = apiModel.zdr_supported;
85
- }
87
+ existing.name = details.display_name || existing.name;
88
+ existing.reasoning = reasoning;
89
+ existing.input = input;
90
+ existing.cost = cost;
91
+ existing.contextWindow = details.context_length || apiModel.max_model_len || existing.contextWindow;
92
+ existing.compat = {
93
+ ...(existing.compat || {}),
94
+ supportsStore: false,
95
+ supportsDeveloperRole: false,
96
+ maxTokensField: 'max_completion_tokens',
97
+ supportsZdr: apiModel.zdr_supported,
98
+ ...(reasoning ? { supportsReasoningEffort: true } : {}),
99
+ };
86
100
  return existing;
87
101
  }
88
102
 
89
- // New model — sensible defaults; curate models.json manually after discovery
103
+ const contextWindow = details.context_length || apiModel.max_model_len || 131072;
90
104
  const model = {
91
105
  id,
92
- name: generateDisplayName(id),
93
- reasoning: false,
94
- input: ['text'],
95
- cost: {
96
- input: 0,
97
- output: 0,
98
- cacheRead: 0,
99
- cacheWrite: 0,
100
- },
101
- contextWindow: apiModel.max_model_len || 131072,
102
- maxTokens: 16384,
106
+ name: details.display_name || generateDisplayName(id),
107
+ reasoning,
108
+ input,
109
+ cost,
110
+ contextWindow,
111
+ maxTokens: contextWindow,
103
112
  compat: {
104
113
  maxTokensField: 'max_completion_tokens',
105
114
  supportsDeveloperRole: false,
106
115
  supportsStore: false,
107
116
  supportsZdr: apiModel.zdr_supported,
117
+ ...(reasoning ? { supportsReasoningEffort: true } : {}),
108
118
  },
109
119
  };
110
120