pi-baseten-provider 1.0.8 → 1.0.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -8
- package/custom-models.json +1 -24
- package/deprecated-models.json +48 -1
- package/index.ts +11 -5
- package/models.json +43 -61
- package/package.json +1 -1
- package/patch.json +184 -18
- package/scripts/update-models.js +116 -11
package/README.md
CHANGED
|
@@ -69,19 +69,18 @@ pi
|
|
|
69
69
|
|
|
70
70
|
| Model | Context | Vision | Reasoning | Input $/M | Output $/M |
|
|
71
71
|
|-------|---------|--------|-----------|-----------|------------|
|
|
72
|
-
|
|
|
72
|
+
| Deepseek V4 Flash 0731 | 1.0M | ❌ | ✅ | $0.13 | $0.26 |
|
|
73
|
+
| Deepseek V4 Pro | 262K | ❌ | ✅ | $1.74 | $3.48 |
|
|
73
74
|
| GLM 4.7 | 200K | ❌ | ✅ | $0.60 | $2.20 |
|
|
74
75
|
| GLM 5.2 | 1.0M | ❌ | ✅ | $1.40 | $4.40 |
|
|
75
76
|
| GLM 5.2 Fast | 524K | ❌ | ✅ | $2.10 | $6.60 |
|
|
76
|
-
| Inkling | 1.0M |
|
|
77
|
-
| Inkling Small | 1.0M |
|
|
78
|
-
| Kimi K2.6 | 262K | ✅ | ✅ | $0.
|
|
79
|
-
| Kimi K2.7 Code | 262K |
|
|
80
|
-
| Kimi K3 | 1.0M |
|
|
81
|
-
| Mercury 2 | 8K | ❌ | ✅ | Free | Free |
|
|
77
|
+
| Inkling | 1.0M | ✅ | ✅ | $1.00 | $4.05 |
|
|
78
|
+
| Inkling Small | 1.0M | ✅ | ✅ | $0.50 | $1.20 |
|
|
79
|
+
| Kimi K2.6 | 262K | ✅ | ✅ | $0.95 | $4.00 |
|
|
80
|
+
| Kimi K2.7 Code | 262K | ✅ | ✅ | $0.95 | $4.00 |
|
|
81
|
+
| Kimi K3 | 1.0M | ✅ | ✅ | $3.00 | $15.00 |
|
|
82
82
|
| Nemotron Ultra | 203K | ❌ | ✅ | $0.60 | $2.40 |
|
|
83
83
|
| OpenAI GPT 120B | 128K | ❌ | ✅ | $0.10 | $0.50 |
|
|
84
|
-
| SID-1 | 33K | ❌ | ✅ | Free | Free |
|
|
85
84
|
|
|
86
85
|
*Costs are per million tokens. Prices subject to change — check [baseten.co/pricing](https://www.baseten.co/pricing/) for current pricing.*
|
|
87
86
|
|
package/custom-models.json
CHANGED
|
@@ -1,24 +1 @@
|
|
|
1
|
-
[
|
|
2
|
-
{
|
|
3
|
-
"id": "deepseek-ai/DeepSeek-V4-Pro",
|
|
4
|
-
"name": "DeepSeek V4 Pro",
|
|
5
|
-
"reasoning": true,
|
|
6
|
-
"input": [
|
|
7
|
-
"text"
|
|
8
|
-
],
|
|
9
|
-
"cost": {
|
|
10
|
-
"input": 1.74,
|
|
11
|
-
"output": 3.48,
|
|
12
|
-
"cacheRead": 0.15,
|
|
13
|
-
"cacheWrite": 0
|
|
14
|
-
},
|
|
15
|
-
"contextWindow": 131000,
|
|
16
|
-
"maxTokens": 131000,
|
|
17
|
-
"compat": {
|
|
18
|
-
"supportsDeveloperRole": true,
|
|
19
|
-
"supportsStore": false,
|
|
20
|
-
"maxTokensField": "max_completion_tokens",
|
|
21
|
-
"thinkingFormat": "openai"
|
|
22
|
-
}
|
|
23
|
-
}
|
|
24
|
-
]
|
|
1
|
+
[]
|
package/deprecated-models.json
CHANGED
|
@@ -1 +1,48 @@
|
|
|
1
|
-
{
|
|
1
|
+
{
|
|
2
|
+
"inception/mercury-2": {
|
|
3
|
+
"id": "inception/mercury-2",
|
|
4
|
+
"name": "Mercury 2",
|
|
5
|
+
"reasoning": true,
|
|
6
|
+
"input": [
|
|
7
|
+
"text"
|
|
8
|
+
],
|
|
9
|
+
"cost": {
|
|
10
|
+
"input": 0,
|
|
11
|
+
"output": 0,
|
|
12
|
+
"cacheRead": 0,
|
|
13
|
+
"cacheWrite": 0
|
|
14
|
+
},
|
|
15
|
+
"contextWindow": 8192,
|
|
16
|
+
"maxTokens": 5000,
|
|
17
|
+
"compat": {
|
|
18
|
+
"supportsDeveloperRole": true,
|
|
19
|
+
"supportsStore": false,
|
|
20
|
+
"maxTokensField": "max_completion_tokens",
|
|
21
|
+
"thinkingFormat": "openai"
|
|
22
|
+
},
|
|
23
|
+
"deprecatedAt": "2026-08-05T02:00:12.114Z"
|
|
24
|
+
},
|
|
25
|
+
"sid/sid-1": {
|
|
26
|
+
"id": "sid/sid-1",
|
|
27
|
+
"name": "SID-1",
|
|
28
|
+
"reasoning": true,
|
|
29
|
+
"input": [
|
|
30
|
+
"text"
|
|
31
|
+
],
|
|
32
|
+
"cost": {
|
|
33
|
+
"input": 0,
|
|
34
|
+
"output": 0,
|
|
35
|
+
"cacheRead": 0,
|
|
36
|
+
"cacheWrite": 0
|
|
37
|
+
},
|
|
38
|
+
"contextWindow": 32768,
|
|
39
|
+
"maxTokens": 5000,
|
|
40
|
+
"compat": {
|
|
41
|
+
"supportsDeveloperRole": true,
|
|
42
|
+
"supportsStore": false,
|
|
43
|
+
"maxTokensField": "max_completion_tokens",
|
|
44
|
+
"thinkingFormat": "openai"
|
|
45
|
+
},
|
|
46
|
+
"deprecatedAt": "2026-08-05T02:00:12.114Z"
|
|
47
|
+
}
|
|
48
|
+
}
|
package/index.ts
CHANGED
|
@@ -49,6 +49,7 @@ interface JsonModel {
|
|
|
49
49
|
contextWindow: number;
|
|
50
50
|
maxTokens: number;
|
|
51
51
|
thinkingLevelMap?: {
|
|
52
|
+
off?: string | null;
|
|
52
53
|
minimal?: string | null;
|
|
53
54
|
low?: string | null;
|
|
54
55
|
medium?: string | null;
|
|
@@ -60,8 +61,10 @@ interface JsonModel {
|
|
|
60
61
|
supportsDeveloperRole?: boolean;
|
|
61
62
|
supportsStore?: boolean;
|
|
62
63
|
maxTokensField?: "max_completion_tokens" | "max_tokens";
|
|
63
|
-
thinkingFormat?: "openai" | "zai" | "qwen" | "qwen-chat-template";
|
|
64
|
+
thinkingFormat?: "openai" | "zai" | "qwen" | "qwen-chat-template" | "chat-template";
|
|
64
65
|
supportsReasoningEffort?: boolean;
|
|
66
|
+
requiresReasoningContentOnAssistantMessages?: boolean;
|
|
67
|
+
chatTemplateKwargs?: Record<string, unknown>;
|
|
65
68
|
};
|
|
66
69
|
}
|
|
67
70
|
|
|
@@ -181,15 +184,18 @@ function transformApiModel(apiModel: any): JsonModel | null {
|
|
|
181
184
|
cost: {
|
|
182
185
|
input: toPerM(pricing.prompt),
|
|
183
186
|
output: toPerM(pricing.completion),
|
|
184
|
-
cacheRead: toPerM(pricing.cache_prompt),
|
|
187
|
+
cacheRead: toPerM(pricing.input_cache_read ?? pricing.cache_prompt),
|
|
185
188
|
cacheWrite: 0,
|
|
186
189
|
},
|
|
187
190
|
contextWindow: apiModel.context_length || 131072,
|
|
188
191
|
maxTokens: apiModel.max_completion_tokens || 131072,
|
|
189
192
|
};
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
+
model.compat = {
|
|
194
|
+
supportsDeveloperRole: true,
|
|
195
|
+
supportsStore: false,
|
|
196
|
+
maxTokensField: "max_completion_tokens",
|
|
197
|
+
...(features.includes("reasoning_effort") ? { supportsReasoningEffort: true } : {}),
|
|
198
|
+
};
|
|
193
199
|
return model;
|
|
194
200
|
}
|
|
195
201
|
|
package/models.json
CHANGED
|
@@ -1,4 +1,26 @@
|
|
|
1
1
|
[
|
|
2
|
+
{
|
|
3
|
+
"id": "deepseek-ai/DeepSeek-V4-Flash-0731",
|
|
4
|
+
"name": "Deepseek V4 Flash 0731",
|
|
5
|
+
"reasoning": true,
|
|
6
|
+
"input": [
|
|
7
|
+
"text"
|
|
8
|
+
],
|
|
9
|
+
"cost": {
|
|
10
|
+
"input": 0.13,
|
|
11
|
+
"output": 0.26,
|
|
12
|
+
"cacheRead": 0.028,
|
|
13
|
+
"cacheWrite": 0
|
|
14
|
+
},
|
|
15
|
+
"contextWindow": 1048576,
|
|
16
|
+
"maxTokens": 1048576,
|
|
17
|
+
"compat": {
|
|
18
|
+
"supportsDeveloperRole": true,
|
|
19
|
+
"supportsStore": false,
|
|
20
|
+
"maxTokensField": "max_completion_tokens",
|
|
21
|
+
"thinkingFormat": "openai"
|
|
22
|
+
}
|
|
23
|
+
},
|
|
2
24
|
{
|
|
3
25
|
"id": "deepseek-ai/DeepSeek-V4-Pro",
|
|
4
26
|
"name": "Deepseek V4 Pro",
|
|
@@ -9,7 +31,7 @@
|
|
|
9
31
|
"cost": {
|
|
10
32
|
"input": 1.74,
|
|
11
33
|
"output": 3.48,
|
|
12
|
-
"cacheRead": 0,
|
|
34
|
+
"cacheRead": 0.145,
|
|
13
35
|
"cacheWrite": 0
|
|
14
36
|
},
|
|
15
37
|
"contextWindow": 262144,
|
|
@@ -31,7 +53,7 @@
|
|
|
31
53
|
"cost": {
|
|
32
54
|
"input": 0.6,
|
|
33
55
|
"output": 2.2,
|
|
34
|
-
"cacheRead": 0,
|
|
56
|
+
"cacheRead": 0.12,
|
|
35
57
|
"cacheWrite": 0
|
|
36
58
|
},
|
|
37
59
|
"contextWindow": 200000,
|
|
@@ -53,7 +75,7 @@
|
|
|
53
75
|
"cost": {
|
|
54
76
|
"input": 1.4,
|
|
55
77
|
"output": 4.4,
|
|
56
|
-
"cacheRead": 0,
|
|
78
|
+
"cacheRead": 0.14,
|
|
57
79
|
"cacheWrite": 0
|
|
58
80
|
},
|
|
59
81
|
"contextWindow": 1048576,
|
|
@@ -75,7 +97,7 @@
|
|
|
75
97
|
"cost": {
|
|
76
98
|
"input": 2.1,
|
|
77
99
|
"output": 6.6,
|
|
78
|
-
"cacheRead": 0,
|
|
100
|
+
"cacheRead": 0.21,
|
|
79
101
|
"cacheWrite": 0
|
|
80
102
|
},
|
|
81
103
|
"contextWindow": 524288,
|
|
@@ -92,12 +114,13 @@
|
|
|
92
114
|
"name": "Inkling",
|
|
93
115
|
"reasoning": true,
|
|
94
116
|
"input": [
|
|
95
|
-
"text"
|
|
117
|
+
"text",
|
|
118
|
+
"image"
|
|
96
119
|
],
|
|
97
120
|
"cost": {
|
|
98
121
|
"input": 1,
|
|
99
122
|
"output": 4.05,
|
|
100
|
-
"cacheRead": 0,
|
|
123
|
+
"cacheRead": 0.17,
|
|
101
124
|
"cacheWrite": 0
|
|
102
125
|
},
|
|
103
126
|
"contextWindow": 1048576,
|
|
@@ -114,12 +137,13 @@
|
|
|
114
137
|
"name": "Inkling Small",
|
|
115
138
|
"reasoning": true,
|
|
116
139
|
"input": [
|
|
117
|
-
"text"
|
|
140
|
+
"text",
|
|
141
|
+
"image"
|
|
118
142
|
],
|
|
119
143
|
"cost": {
|
|
120
|
-
"input": 0,
|
|
121
|
-
"output":
|
|
122
|
-
"cacheRead": 0,
|
|
144
|
+
"input": 0.5,
|
|
145
|
+
"output": 1.2,
|
|
146
|
+
"cacheRead": 0.1,
|
|
123
147
|
"cacheWrite": 0
|
|
124
148
|
},
|
|
125
149
|
"contextWindow": 1048576,
|
|
@@ -142,7 +166,7 @@
|
|
|
142
166
|
"cost": {
|
|
143
167
|
"input": 0.95,
|
|
144
168
|
"output": 4,
|
|
145
|
-
"cacheRead": 0,
|
|
169
|
+
"cacheRead": 0.16,
|
|
146
170
|
"cacheWrite": 0
|
|
147
171
|
},
|
|
148
172
|
"contextWindow": 262000,
|
|
@@ -159,12 +183,13 @@
|
|
|
159
183
|
"name": "Kimi K2.7 Code",
|
|
160
184
|
"reasoning": true,
|
|
161
185
|
"input": [
|
|
162
|
-
"text"
|
|
186
|
+
"text",
|
|
187
|
+
"image"
|
|
163
188
|
],
|
|
164
189
|
"cost": {
|
|
165
190
|
"input": 0.95,
|
|
166
191
|
"output": 4,
|
|
167
|
-
"cacheRead": 0,
|
|
192
|
+
"cacheRead": 0.16,
|
|
168
193
|
"cacheWrite": 0
|
|
169
194
|
},
|
|
170
195
|
"contextWindow": 262000,
|
|
@@ -181,12 +206,13 @@
|
|
|
181
206
|
"name": "Kimi K3",
|
|
182
207
|
"reasoning": true,
|
|
183
208
|
"input": [
|
|
184
|
-
"text"
|
|
209
|
+
"text",
|
|
210
|
+
"image"
|
|
185
211
|
],
|
|
186
212
|
"cost": {
|
|
187
213
|
"input": 3,
|
|
188
214
|
"output": 15,
|
|
189
|
-
"cacheRead": 0,
|
|
215
|
+
"cacheRead": 0.3,
|
|
190
216
|
"cacheWrite": 0
|
|
191
217
|
},
|
|
192
218
|
"contextWindow": 1048576,
|
|
@@ -198,28 +224,6 @@
|
|
|
198
224
|
"thinkingFormat": "openai"
|
|
199
225
|
}
|
|
200
226
|
},
|
|
201
|
-
{
|
|
202
|
-
"id": "inception/mercury-2",
|
|
203
|
-
"name": "Mercury 2",
|
|
204
|
-
"reasoning": true,
|
|
205
|
-
"input": [
|
|
206
|
-
"text"
|
|
207
|
-
],
|
|
208
|
-
"cost": {
|
|
209
|
-
"input": 0,
|
|
210
|
-
"output": 0,
|
|
211
|
-
"cacheRead": 0,
|
|
212
|
-
"cacheWrite": 0
|
|
213
|
-
},
|
|
214
|
-
"contextWindow": 8192,
|
|
215
|
-
"maxTokens": 5000,
|
|
216
|
-
"compat": {
|
|
217
|
-
"supportsDeveloperRole": true,
|
|
218
|
-
"supportsStore": false,
|
|
219
|
-
"maxTokensField": "max_completion_tokens",
|
|
220
|
-
"thinkingFormat": "openai"
|
|
221
|
-
}
|
|
222
|
-
},
|
|
223
227
|
{
|
|
224
228
|
"id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B",
|
|
225
229
|
"name": "Nemotron Ultra",
|
|
@@ -230,7 +234,7 @@
|
|
|
230
234
|
"cost": {
|
|
231
235
|
"input": 0.6,
|
|
232
236
|
"output": 2.4,
|
|
233
|
-
"cacheRead": 0,
|
|
237
|
+
"cacheRead": 0.12,
|
|
234
238
|
"cacheWrite": 0
|
|
235
239
|
},
|
|
236
240
|
"contextWindow": 202800,
|
|
@@ -252,7 +256,7 @@
|
|
|
252
256
|
"cost": {
|
|
253
257
|
"input": 0.1,
|
|
254
258
|
"output": 0.5,
|
|
255
|
-
"cacheRead": 0,
|
|
259
|
+
"cacheRead": 0.1,
|
|
256
260
|
"cacheWrite": 0
|
|
257
261
|
},
|
|
258
262
|
"contextWindow": 128072,
|
|
@@ -264,27 +268,5 @@
|
|
|
264
268
|
"thinkingFormat": "openai",
|
|
265
269
|
"supportsReasoningEffort": true
|
|
266
270
|
}
|
|
267
|
-
},
|
|
268
|
-
{
|
|
269
|
-
"id": "sid/sid-1",
|
|
270
|
-
"name": "SID-1",
|
|
271
|
-
"reasoning": true,
|
|
272
|
-
"input": [
|
|
273
|
-
"text"
|
|
274
|
-
],
|
|
275
|
-
"cost": {
|
|
276
|
-
"input": 0,
|
|
277
|
-
"output": 0,
|
|
278
|
-
"cacheRead": 0,
|
|
279
|
-
"cacheWrite": 0
|
|
280
|
-
},
|
|
281
|
-
"contextWindow": 32768,
|
|
282
|
-
"maxTokens": 5000,
|
|
283
|
-
"compat": {
|
|
284
|
-
"supportsDeveloperRole": true,
|
|
285
|
-
"supportsStore": false,
|
|
286
|
-
"maxTokensField": "max_completion_tokens",
|
|
287
|
-
"thinkingFormat": "openai"
|
|
288
|
-
}
|
|
289
271
|
}
|
|
290
272
|
]
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-baseten-provider",
|
|
3
|
-
"version": "1.0.
|
|
3
|
+
"version": "1.0.10",
|
|
4
4
|
"description": "Baseten provider extension for pi - Access DeepSeek, Kimi, GLM, MiniMax, Nemotron, and GPT-OSS models through the Baseten Model API",
|
|
5
5
|
"author": "monotykamary",
|
|
6
6
|
"homepage": "https://github.com/monotykamary/pi-baseten-provider#readme",
|
package/patch.json
CHANGED
|
@@ -1,54 +1,220 @@
|
|
|
1
1
|
{
|
|
2
|
-
"
|
|
3
|
-
"
|
|
4
|
-
"
|
|
5
|
-
"
|
|
6
|
-
"
|
|
2
|
+
"deepseek-ai/DeepSeek-V4-Flash-0731": {
|
|
3
|
+
"maxTokens": 384000,
|
|
4
|
+
"thinkingLevelMap": {
|
|
5
|
+
"off": "none",
|
|
6
|
+
"minimal": null,
|
|
7
|
+
"low": "low",
|
|
8
|
+
"medium": null,
|
|
9
|
+
"high": "high",
|
|
10
|
+
"xhigh": null,
|
|
11
|
+
"max": "max"
|
|
7
12
|
},
|
|
8
13
|
"compat": {
|
|
9
|
-
"thinkingFormat": "
|
|
14
|
+
"thinkingFormat": "chat-template",
|
|
15
|
+
"supportsReasoningEffort": false,
|
|
16
|
+
"requiresReasoningContentOnAssistantMessages": true,
|
|
17
|
+
"chatTemplateKwargs": {
|
|
18
|
+
"thinking": {
|
|
19
|
+
"$var": "thinking.enabled"
|
|
20
|
+
},
|
|
21
|
+
"reasoning_effort": {
|
|
22
|
+
"$var": "thinking.effort"
|
|
23
|
+
}
|
|
24
|
+
}
|
|
10
25
|
}
|
|
11
26
|
},
|
|
12
|
-
"
|
|
13
|
-
"
|
|
27
|
+
"deepseek-ai/DeepSeek-V4-Pro": {
|
|
28
|
+
"thinkingLevelMap": {
|
|
29
|
+
"off": "none",
|
|
30
|
+
"minimal": "minimal",
|
|
31
|
+
"low": "low",
|
|
32
|
+
"medium": "medium",
|
|
33
|
+
"high": "high",
|
|
34
|
+
"xhigh": "xhigh",
|
|
35
|
+
"max": "max"
|
|
36
|
+
},
|
|
14
37
|
"compat": {
|
|
15
|
-
"thinkingFormat": "
|
|
38
|
+
"thinkingFormat": "openai",
|
|
39
|
+
"supportsReasoningEffort": true,
|
|
40
|
+
"requiresReasoningContentOnAssistantMessages": true
|
|
16
41
|
}
|
|
17
42
|
},
|
|
18
43
|
"zai-org/GLM-4.7": {
|
|
19
44
|
"reasoning": true,
|
|
45
|
+
"thinkingLevelMap": {
|
|
46
|
+
"off": "none",
|
|
47
|
+
"minimal": null,
|
|
48
|
+
"low": null,
|
|
49
|
+
"medium": null,
|
|
50
|
+
"high": "high",
|
|
51
|
+
"xhigh": null,
|
|
52
|
+
"max": null
|
|
53
|
+
},
|
|
20
54
|
"compat": {
|
|
21
|
-
"thinkingFormat": "qwen-chat-template"
|
|
55
|
+
"thinkingFormat": "qwen-chat-template",
|
|
56
|
+
"supportsReasoningEffort": false,
|
|
57
|
+
"requiresReasoningContentOnAssistantMessages": true
|
|
22
58
|
}
|
|
23
59
|
},
|
|
24
|
-
"zai-org/GLM-5": {
|
|
25
|
-
"
|
|
60
|
+
"zai-org/GLM-5.2": {
|
|
61
|
+
"thinkingLevelMap": {
|
|
62
|
+
"off": "none",
|
|
63
|
+
"minimal": null,
|
|
64
|
+
"low": null,
|
|
65
|
+
"medium": null,
|
|
66
|
+
"high": "high",
|
|
67
|
+
"xhigh": null,
|
|
68
|
+
"max": "max"
|
|
69
|
+
},
|
|
26
70
|
"compat": {
|
|
27
|
-
"thinkingFormat": "
|
|
71
|
+
"thinkingFormat": "openai",
|
|
72
|
+
"supportsReasoningEffort": true,
|
|
73
|
+
"requiresReasoningContentOnAssistantMessages": true
|
|
28
74
|
}
|
|
29
75
|
},
|
|
30
|
-
"zai-org/GLM-5.2": {
|
|
76
|
+
"zai-org/GLM-5.2-Fast": {
|
|
77
|
+
"thinkingLevelMap": {
|
|
78
|
+
"off": "none",
|
|
79
|
+
"minimal": null,
|
|
80
|
+
"low": null,
|
|
81
|
+
"medium": null,
|
|
82
|
+
"high": "high",
|
|
83
|
+
"xhigh": null,
|
|
84
|
+
"max": "max"
|
|
85
|
+
},
|
|
31
86
|
"compat": {
|
|
32
|
-
"thinkingFormat": "
|
|
87
|
+
"thinkingFormat": "openai",
|
|
88
|
+
"supportsReasoningEffort": true,
|
|
89
|
+
"requiresReasoningContentOnAssistantMessages": true
|
|
90
|
+
}
|
|
91
|
+
},
|
|
92
|
+
"thinkingmachines/inkling": {
|
|
93
|
+
"thinkingLevelMap": {
|
|
94
|
+
"off": "none",
|
|
95
|
+
"minimal": "minimal",
|
|
96
|
+
"low": "low",
|
|
97
|
+
"medium": "medium",
|
|
98
|
+
"high": "high",
|
|
99
|
+
"xhigh": "xhigh",
|
|
100
|
+
"max": "max"
|
|
33
101
|
},
|
|
102
|
+
"compat": {
|
|
103
|
+
"thinkingFormat": "openai",
|
|
104
|
+
"supportsReasoningEffort": true
|
|
105
|
+
}
|
|
106
|
+
},
|
|
107
|
+
"thinkingmachines/inkling-small": {
|
|
34
108
|
"thinkingLevelMap": {
|
|
109
|
+
"off": "none",
|
|
35
110
|
"minimal": "minimal",
|
|
36
111
|
"low": "low",
|
|
37
112
|
"medium": "medium",
|
|
38
113
|
"high": "high",
|
|
114
|
+
"xhigh": "xhigh",
|
|
39
115
|
"max": "max"
|
|
116
|
+
},
|
|
117
|
+
"compat": {
|
|
118
|
+
"thinkingFormat": "openai",
|
|
119
|
+
"supportsReasoningEffort": true
|
|
120
|
+
}
|
|
121
|
+
},
|
|
122
|
+
"moonshotai/Kimi-K2.6": {
|
|
123
|
+
"thinkingLevelMap": {
|
|
124
|
+
"off": "none",
|
|
125
|
+
"minimal": null,
|
|
126
|
+
"low": null,
|
|
127
|
+
"medium": null,
|
|
128
|
+
"high": "high",
|
|
129
|
+
"xhigh": null,
|
|
130
|
+
"max": null
|
|
131
|
+
},
|
|
132
|
+
"compat": {
|
|
133
|
+
"thinkingFormat": "qwen-chat-template",
|
|
134
|
+
"supportsReasoningEffort": false,
|
|
135
|
+
"requiresReasoningContentOnAssistantMessages": true
|
|
40
136
|
}
|
|
41
137
|
},
|
|
42
138
|
"moonshotai/Kimi-K2.7-Code": {
|
|
43
139
|
"thinkingLevelMap": {
|
|
140
|
+
"off": "none",
|
|
141
|
+
"minimal": null,
|
|
142
|
+
"low": null,
|
|
143
|
+
"medium": null,
|
|
144
|
+
"high": "high",
|
|
145
|
+
"xhigh": null,
|
|
146
|
+
"max": null
|
|
147
|
+
},
|
|
148
|
+
"compat": {
|
|
149
|
+
"thinkingFormat": "qwen-chat-template",
|
|
150
|
+
"supportsReasoningEffort": false,
|
|
151
|
+
"requiresReasoningContentOnAssistantMessages": true
|
|
152
|
+
}
|
|
153
|
+
},
|
|
154
|
+
"moonshotai/Kimi-K3": {
|
|
155
|
+
"thinkingLevelMap": {
|
|
156
|
+
"off": "none",
|
|
157
|
+
"minimal": null,
|
|
158
|
+
"low": "low",
|
|
159
|
+
"medium": null,
|
|
160
|
+
"high": "high",
|
|
161
|
+
"xhigh": null,
|
|
162
|
+
"max": "max"
|
|
163
|
+
},
|
|
164
|
+
"compat": {
|
|
165
|
+
"thinkingFormat": "openai",
|
|
166
|
+
"supportsReasoningEffort": true,
|
|
167
|
+
"requiresReasoningContentOnAssistantMessages": true
|
|
168
|
+
}
|
|
169
|
+
},
|
|
170
|
+
"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B": {
|
|
171
|
+
"thinkingLevelMap": {
|
|
172
|
+
"off": "none",
|
|
44
173
|
"minimal": null,
|
|
174
|
+
"low": null,
|
|
175
|
+
"medium": null,
|
|
176
|
+
"high": "high",
|
|
177
|
+
"xhigh": null,
|
|
178
|
+
"max": null
|
|
179
|
+
},
|
|
180
|
+
"compat": {
|
|
181
|
+
"thinkingFormat": "qwen-chat-template",
|
|
182
|
+
"supportsReasoningEffort": false,
|
|
183
|
+
"requiresReasoningContentOnAssistantMessages": true
|
|
184
|
+
}
|
|
185
|
+
},
|
|
186
|
+
"openai/gpt-oss-120b": {
|
|
187
|
+
"thinkingLevelMap": {
|
|
188
|
+
"off": "none",
|
|
189
|
+
"minimal": "minimal",
|
|
190
|
+
"low": "low",
|
|
191
|
+
"medium": "medium",
|
|
192
|
+
"high": "high",
|
|
193
|
+
"xhigh": "xhigh",
|
|
194
|
+
"max": "max"
|
|
195
|
+
},
|
|
196
|
+
"compat": {
|
|
197
|
+
"thinkingFormat": "openai",
|
|
198
|
+
"supportsReasoningEffort": true
|
|
199
|
+
}
|
|
200
|
+
},
|
|
201
|
+
"inception/mercury-2": {
|
|
202
|
+
"reasoning": true,
|
|
203
|
+
"thinkingLevelMap": {
|
|
204
|
+
"off": "instant",
|
|
205
|
+
"minimal": "low",
|
|
45
206
|
"low": "low",
|
|
46
207
|
"medium": "medium",
|
|
47
208
|
"high": "high",
|
|
48
|
-
"xhigh": null
|
|
209
|
+
"xhigh": null,
|
|
210
|
+
"max": null
|
|
211
|
+
},
|
|
212
|
+
"compat": {
|
|
213
|
+
"thinkingFormat": "openai",
|
|
214
|
+
"supportsReasoningEffort": true
|
|
49
215
|
}
|
|
50
216
|
},
|
|
51
|
-
"
|
|
52
|
-
"reasoning":
|
|
217
|
+
"sid/sid-1": {
|
|
218
|
+
"reasoning": false
|
|
53
219
|
}
|
|
54
220
|
}
|
package/scripts/update-models.js
CHANGED
|
@@ -17,15 +17,119 @@
|
|
|
17
17
|
* patch.json and custom-models.json are applied at runtime by the provider.
|
|
18
18
|
* They are NOT baked into models.json, but ARE used to generate the README table.
|
|
19
19
|
*
|
|
20
|
-
*
|
|
20
|
+
* API key: the stored `baseten` credential in ~/.pi/agent/auth.json wins, then
|
|
21
|
+
* the BASETEN_API_KEY environment variable. The script refuses to run without one.
|
|
21
22
|
*/
|
|
22
23
|
|
|
23
24
|
import fs from 'fs';
|
|
25
|
+
import os from 'os';
|
|
26
|
+
import { execSync } from 'child_process';
|
|
24
27
|
import path from 'path';
|
|
25
28
|
import { fileURLToPath } from 'url';
|
|
26
29
|
|
|
27
30
|
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
|
28
31
|
|
|
32
|
+
// pi's agent directory: PI_CODING_AGENT_DIR (with ~ expansion) or ~/.pi/agent.
|
|
33
|
+
function piAgentDir() {
|
|
34
|
+
const envDir = process.env.PI_CODING_AGENT_DIR;
|
|
35
|
+
if (envDir) {
|
|
36
|
+
return envDir.startsWith('~/') || envDir === '~'
|
|
37
|
+
? path.join(os.homedir(), envDir.slice(1))
|
|
38
|
+
: envDir;
|
|
39
|
+
}
|
|
40
|
+
return path.join(os.homedir(), '.pi', 'agent');
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
const AUTH_JSON_PATH = path.join(piAgentDir(), 'auth.json');
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* Resolve a configured value using pi's semantics (resolve-config-value.ts in
|
|
47
|
+
* pi-mono): "!command" runs via the shell (10s timeout) and uses trimmed
|
|
48
|
+
* stdout; "$VAR" / "${VAR}" interpolate environment variables ("$$" escapes a
|
|
49
|
+
* literal "$", "$!" a literal "!"); anything else is a literal. Returns
|
|
50
|
+
* undefined when a referenced env var is unset or a command fails.
|
|
51
|
+
*/
|
|
52
|
+
function resolveConfigValue(config, env) {
|
|
53
|
+
if (typeof config !== 'string' || config.length === 0) return undefined;
|
|
54
|
+
if (config.startsWith('!')) {
|
|
55
|
+
try {
|
|
56
|
+
const out = execSync(config.slice(1), {
|
|
57
|
+
encoding: 'utf8',
|
|
58
|
+
timeout: 10000,
|
|
59
|
+
stdio: ['ignore', 'pipe', 'ignore'],
|
|
60
|
+
});
|
|
61
|
+
return out.trim() || undefined;
|
|
62
|
+
} catch {
|
|
63
|
+
return undefined;
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
const ENV_NAME_RE = /^[A-Za-z_][A-Za-z0-9_]*$/;
|
|
67
|
+
let resolved = '';
|
|
68
|
+
let index = 0;
|
|
69
|
+
while (index < config.length) {
|
|
70
|
+
const dollar = config.indexOf('$', index);
|
|
71
|
+
if (dollar < 0) {
|
|
72
|
+
resolved += config.slice(index);
|
|
73
|
+
break;
|
|
74
|
+
}
|
|
75
|
+
resolved += config.slice(index, dollar);
|
|
76
|
+
const next = config[dollar + 1];
|
|
77
|
+
let name;
|
|
78
|
+
if (next === '$' || next === '!') {
|
|
79
|
+
resolved += next;
|
|
80
|
+
index = dollar + 2;
|
|
81
|
+
continue;
|
|
82
|
+
} else if (next === '{') {
|
|
83
|
+
const end = config.indexOf('}', dollar + 2);
|
|
84
|
+
if (end < 0) {
|
|
85
|
+
resolved += '$';
|
|
86
|
+
index = dollar + 1;
|
|
87
|
+
continue;
|
|
88
|
+
}
|
|
89
|
+
const inner = config.slice(dollar + 2, end);
|
|
90
|
+
if (!ENV_NAME_RE.test(inner)) {
|
|
91
|
+
resolved += config.slice(dollar, end + 1);
|
|
92
|
+
index = end + 1;
|
|
93
|
+
continue;
|
|
94
|
+
}
|
|
95
|
+
name = inner;
|
|
96
|
+
index = end + 1;
|
|
97
|
+
} else {
|
|
98
|
+
const match = config.slice(dollar + 1).match(/^[A-Za-z_][A-Za-z0-9_]*/);
|
|
99
|
+
if (!match) {
|
|
100
|
+
resolved += '$';
|
|
101
|
+
index = dollar + 1;
|
|
102
|
+
continue;
|
|
103
|
+
}
|
|
104
|
+
name = match[0];
|
|
105
|
+
index = dollar + 1 + name.length;
|
|
106
|
+
}
|
|
107
|
+
const value = (env && env[name]) || process.env[name] || undefined;
|
|
108
|
+
if (value === undefined) return undefined;
|
|
109
|
+
resolved += value;
|
|
110
|
+
}
|
|
111
|
+
return resolved;
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
/**
|
|
115
|
+
* The API key, resolved the way pi itself resolves it for this provider: the
|
|
116
|
+
* stored `baseten` credential in ~/.pi/agent/auth.json wins, then
|
|
117
|
+
* the BASETEN_API_KEY environment variable.
|
|
118
|
+
*/
|
|
119
|
+
function resolveApiKey() {
|
|
120
|
+
try {
|
|
121
|
+
const auth = JSON.parse(fs.readFileSync(AUTH_JSON_PATH, 'utf8'));
|
|
122
|
+
const credential = auth?.baseten;
|
|
123
|
+
if (credential && credential.type === 'api_key' && typeof credential.key === 'string') {
|
|
124
|
+
const key = resolveConfigValue(credential.key, credential.env);
|
|
125
|
+
if (key) return key;
|
|
126
|
+
}
|
|
127
|
+
} catch {
|
|
128
|
+
// Missing or unparseable auth.json: fall through to the env var.
|
|
129
|
+
}
|
|
130
|
+
return process.env.BASETEN_API_KEY || undefined;
|
|
131
|
+
}
|
|
132
|
+
|
|
29
133
|
const MODELS_API_URL = 'https://inference.baseten.co/v1/models';
|
|
30
134
|
const MODELS_JSON_PATH = path.join(__dirname, '..', 'models.json');
|
|
31
135
|
const PATCH_JSON_PATH = path.join(__dirname, '..', 'patch.json');
|
|
@@ -58,9 +162,9 @@ function toPerMillion(val) {
|
|
|
58
162
|
// ─── API fetch ───────────────────────────────────────────────────────────────
|
|
59
163
|
|
|
60
164
|
async function fetchModels() {
|
|
61
|
-
const apiKey =
|
|
165
|
+
const apiKey = resolveApiKey();
|
|
62
166
|
if (!apiKey) {
|
|
63
|
-
throw new Error('
|
|
167
|
+
throw new Error('No API key found: no `baseten` credential resolved from ' + AUTH_JSON_PATH + ' and BASETEN_API_KEY is not set');
|
|
64
168
|
}
|
|
65
169
|
|
|
66
170
|
console.log(`Fetching models from ${MODELS_API_URL}...`);
|
|
@@ -96,16 +200,17 @@ function transformApiModel(apiModel, existingModelsMap) {
|
|
|
96
200
|
}
|
|
97
201
|
// Update features from API
|
|
98
202
|
const features = apiModel.supported_features || [];
|
|
99
|
-
existing.reasoning = features.includes('reasoning')
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
}
|
|
203
|
+
existing.reasoning = features.includes('reasoning');
|
|
204
|
+
const hasVision = (apiModel.input_modalities || []).includes('image');
|
|
205
|
+
existing.input = hasVision ? ['text', 'image'] : ['text'];
|
|
103
206
|
// Update pricing from API
|
|
104
207
|
const pricing = apiModel.pricing || {};
|
|
105
208
|
const inputCost = toPerMillion(pricing.prompt);
|
|
106
209
|
const outputCost = toPerMillion(pricing.completion);
|
|
107
|
-
|
|
108
|
-
if (
|
|
210
|
+
const cacheReadCost = toPerMillion(pricing.input_cache_read ?? pricing.cache_prompt);
|
|
211
|
+
if (inputCost !== null) existing.cost.input = inputCost;
|
|
212
|
+
if (outputCost !== null) existing.cost.output = outputCost;
|
|
213
|
+
if (cacheReadCost !== null) existing.cost.cacheRead = cacheReadCost;
|
|
109
214
|
return existing;
|
|
110
215
|
}
|
|
111
216
|
|
|
@@ -113,7 +218,7 @@ function transformApiModel(apiModel, existingModelsMap) {
|
|
|
113
218
|
const features = apiModel.supported_features || [];
|
|
114
219
|
const pricing = apiModel.pricing || {};
|
|
115
220
|
const hasReasoning = features.includes('reasoning');
|
|
116
|
-
const hasVision =
|
|
221
|
+
const hasVision = (apiModel.input_modalities || []).includes('image');
|
|
117
222
|
|
|
118
223
|
const inputTypes = ['text'];
|
|
119
224
|
if (hasVision) inputTypes.push('image');
|
|
@@ -130,7 +235,7 @@ function transformApiModel(apiModel, existingModelsMap) {
|
|
|
130
235
|
cost: {
|
|
131
236
|
input: inputCost,
|
|
132
237
|
output: outputCost,
|
|
133
|
-
cacheRead: 0,
|
|
238
|
+
cacheRead: toPerMillion(pricing.input_cache_read ?? pricing.cache_prompt) || 0,
|
|
134
239
|
cacheWrite: 0,
|
|
135
240
|
},
|
|
136
241
|
contextWindow: apiModel.context_length || 131072,
|