pi-baseten-provider 1.0.8 → 1.0.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -69,19 +69,18 @@ pi
69
69
 
70
70
  | Model | Context | Vision | Reasoning | Input $/M | Output $/M |
71
71
  |-------|---------|--------|-----------|-----------|------------|
72
- | DeepSeek V4 Pro | 131K | ❌ | ✅ | $1.74 | $3.48 |
72
+ | Deepseek V4 Flash 0731 | 1.0M | ❌ | ✅ | $0.13 | $0.26 |
73
+ | Deepseek V4 Pro | 262K | ❌ | ✅ | $1.74 | $3.48 |
73
74
  | GLM 4.7 | 200K | ❌ | ✅ | $0.60 | $2.20 |
74
75
  | GLM 5.2 | 1.0M | ❌ | ✅ | $1.40 | $4.40 |
75
76
  | GLM 5.2 Fast | 524K | ❌ | ✅ | $2.10 | $6.60 |
76
- | Inkling | 1.0M | ❌ | ✅ | $1.00 | $4.05 |
77
- | Inkling Small | 1.0M | ❌ | ✅ | Free | Free |
78
- | Kimi K2.6 | 262K | ✅ | ✅ | $0.60 | $3.00 |
79
- | Kimi K2.7 Code | 262K | ❌ | ✅ | $0.95 | $4.00 |
80
- | Kimi K3 | 1.0M | ❌ | ✅ | $3.00 | $15.00 |
81
- | Mercury 2 | 8K | ❌ | ✅ | Free | Free |
77
+ | Inkling | 1.0M | ✅ | ✅ | $1.00 | $4.05 |
78
+ | Inkling Small | 1.0M | ✅ | ✅ | $0.50 | $1.20 |
79
+ | Kimi K2.6 | 262K | ✅ | ✅ | $0.95 | $4.00 |
80
+ | Kimi K2.7 Code | 262K | ✅ | ✅ | $0.95 | $4.00 |
81
+ | Kimi K3 | 1.0M | ✅ | ✅ | $3.00 | $15.00 |
82
82
  | Nemotron Ultra | 203K | ❌ | ✅ | $0.60 | $2.40 |
83
83
  | OpenAI GPT 120B | 128K | ❌ | ✅ | $0.10 | $0.50 |
84
- | SID-1 | 33K | ❌ | ✅ | Free | Free |
85
84
 
86
85
  *Costs are per million tokens. Prices subject to change — check [baseten.co/pricing](https://www.baseten.co/pricing/) for current pricing.*
87
86
 
@@ -1,24 +1 @@
1
- [
2
- {
3
- "id": "deepseek-ai/DeepSeek-V4-Pro",
4
- "name": "DeepSeek V4 Pro",
5
- "reasoning": true,
6
- "input": [
7
- "text"
8
- ],
9
- "cost": {
10
- "input": 1.74,
11
- "output": 3.48,
12
- "cacheRead": 0.15,
13
- "cacheWrite": 0
14
- },
15
- "contextWindow": 131000,
16
- "maxTokens": 131000,
17
- "compat": {
18
- "supportsDeveloperRole": true,
19
- "supportsStore": false,
20
- "maxTokensField": "max_completion_tokens",
21
- "thinkingFormat": "openai"
22
- }
23
- }
24
- ]
1
+ []
@@ -1 +1,48 @@
1
- {}
1
+ {
2
+ "inception/mercury-2": {
3
+ "id": "inception/mercury-2",
4
+ "name": "Mercury 2",
5
+ "reasoning": true,
6
+ "input": [
7
+ "text"
8
+ ],
9
+ "cost": {
10
+ "input": 0,
11
+ "output": 0,
12
+ "cacheRead": 0,
13
+ "cacheWrite": 0
14
+ },
15
+ "contextWindow": 8192,
16
+ "maxTokens": 5000,
17
+ "compat": {
18
+ "supportsDeveloperRole": true,
19
+ "supportsStore": false,
20
+ "maxTokensField": "max_completion_tokens",
21
+ "thinkingFormat": "openai"
22
+ },
23
+ "deprecatedAt": "2026-08-05T02:00:12.114Z"
24
+ },
25
+ "sid/sid-1": {
26
+ "id": "sid/sid-1",
27
+ "name": "SID-1",
28
+ "reasoning": true,
29
+ "input": [
30
+ "text"
31
+ ],
32
+ "cost": {
33
+ "input": 0,
34
+ "output": 0,
35
+ "cacheRead": 0,
36
+ "cacheWrite": 0
37
+ },
38
+ "contextWindow": 32768,
39
+ "maxTokens": 5000,
40
+ "compat": {
41
+ "supportsDeveloperRole": true,
42
+ "supportsStore": false,
43
+ "maxTokensField": "max_completion_tokens",
44
+ "thinkingFormat": "openai"
45
+ },
46
+ "deprecatedAt": "2026-08-05T02:00:12.114Z"
47
+ }
48
+ }
package/index.ts CHANGED
@@ -49,6 +49,7 @@ interface JsonModel {
49
49
  contextWindow: number;
50
50
  maxTokens: number;
51
51
  thinkingLevelMap?: {
52
+ off?: string | null;
52
53
  minimal?: string | null;
53
54
  low?: string | null;
54
55
  medium?: string | null;
@@ -60,8 +61,10 @@ interface JsonModel {
60
61
  supportsDeveloperRole?: boolean;
61
62
  supportsStore?: boolean;
62
63
  maxTokensField?: "max_completion_tokens" | "max_tokens";
63
- thinkingFormat?: "openai" | "zai" | "qwen" | "qwen-chat-template";
64
+ thinkingFormat?: "openai" | "zai" | "qwen" | "qwen-chat-template" | "chat-template";
64
65
  supportsReasoningEffort?: boolean;
66
+ requiresReasoningContentOnAssistantMessages?: boolean;
67
+ chatTemplateKwargs?: Record<string, unknown>;
65
68
  };
66
69
  }
67
70
 
@@ -181,15 +184,18 @@ function transformApiModel(apiModel: any): JsonModel | null {
181
184
  cost: {
182
185
  input: toPerM(pricing.prompt),
183
186
  output: toPerM(pricing.completion),
184
- cacheRead: toPerM(pricing.cache_prompt),
187
+ cacheRead: toPerM(pricing.input_cache_read ?? pricing.cache_prompt),
185
188
  cacheWrite: 0,
186
189
  },
187
190
  contextWindow: apiModel.context_length || 131072,
188
191
  maxTokens: apiModel.max_completion_tokens || 131072,
189
192
  };
190
- if (features.includes("reasoning_effort")) {
191
- model.compat = { ...model.compat, supportsReasoningEffort: true };
192
- }
193
+ model.compat = {
194
+ supportsDeveloperRole: true,
195
+ supportsStore: false,
196
+ maxTokensField: "max_completion_tokens",
197
+ ...(features.includes("reasoning_effort") ? { supportsReasoningEffort: true } : {}),
198
+ };
193
199
  return model;
194
200
  }
195
201
 
package/models.json CHANGED
@@ -1,4 +1,26 @@
1
1
  [
2
+ {
3
+ "id": "deepseek-ai/DeepSeek-V4-Flash-0731",
4
+ "name": "Deepseek V4 Flash 0731",
5
+ "reasoning": true,
6
+ "input": [
7
+ "text"
8
+ ],
9
+ "cost": {
10
+ "input": 0.13,
11
+ "output": 0.26,
12
+ "cacheRead": 0.028,
13
+ "cacheWrite": 0
14
+ },
15
+ "contextWindow": 1048576,
16
+ "maxTokens": 1048576,
17
+ "compat": {
18
+ "supportsDeveloperRole": true,
19
+ "supportsStore": false,
20
+ "maxTokensField": "max_completion_tokens",
21
+ "thinkingFormat": "openai"
22
+ }
23
+ },
2
24
  {
3
25
  "id": "deepseek-ai/DeepSeek-V4-Pro",
4
26
  "name": "Deepseek V4 Pro",
@@ -9,7 +31,7 @@
9
31
  "cost": {
10
32
  "input": 1.74,
11
33
  "output": 3.48,
12
- "cacheRead": 0,
34
+ "cacheRead": 0.145,
13
35
  "cacheWrite": 0
14
36
  },
15
37
  "contextWindow": 262144,
@@ -31,7 +53,7 @@
31
53
  "cost": {
32
54
  "input": 0.6,
33
55
  "output": 2.2,
34
- "cacheRead": 0,
56
+ "cacheRead": 0.12,
35
57
  "cacheWrite": 0
36
58
  },
37
59
  "contextWindow": 200000,
@@ -53,7 +75,7 @@
53
75
  "cost": {
54
76
  "input": 1.4,
55
77
  "output": 4.4,
56
- "cacheRead": 0,
78
+ "cacheRead": 0.14,
57
79
  "cacheWrite": 0
58
80
  },
59
81
  "contextWindow": 1048576,
@@ -75,7 +97,7 @@
75
97
  "cost": {
76
98
  "input": 2.1,
77
99
  "output": 6.6,
78
- "cacheRead": 0,
100
+ "cacheRead": 0.21,
79
101
  "cacheWrite": 0
80
102
  },
81
103
  "contextWindow": 524288,
@@ -92,12 +114,13 @@
92
114
  "name": "Inkling",
93
115
  "reasoning": true,
94
116
  "input": [
95
- "text"
117
+ "text",
118
+ "image"
96
119
  ],
97
120
  "cost": {
98
121
  "input": 1,
99
122
  "output": 4.05,
100
- "cacheRead": 0,
123
+ "cacheRead": 0.17,
101
124
  "cacheWrite": 0
102
125
  },
103
126
  "contextWindow": 1048576,
@@ -114,12 +137,13 @@
114
137
  "name": "Inkling Small",
115
138
  "reasoning": true,
116
139
  "input": [
117
- "text"
140
+ "text",
141
+ "image"
118
142
  ],
119
143
  "cost": {
120
- "input": 0,
121
- "output": 0,
122
- "cacheRead": 0,
144
+ "input": 0.5,
145
+ "output": 1.2,
146
+ "cacheRead": 0.1,
123
147
  "cacheWrite": 0
124
148
  },
125
149
  "contextWindow": 1048576,
@@ -142,7 +166,7 @@
142
166
  "cost": {
143
167
  "input": 0.95,
144
168
  "output": 4,
145
- "cacheRead": 0,
169
+ "cacheRead": 0.16,
146
170
  "cacheWrite": 0
147
171
  },
148
172
  "contextWindow": 262000,
@@ -159,12 +183,13 @@
159
183
  "name": "Kimi K2.7 Code",
160
184
  "reasoning": true,
161
185
  "input": [
162
- "text"
186
+ "text",
187
+ "image"
163
188
  ],
164
189
  "cost": {
165
190
  "input": 0.95,
166
191
  "output": 4,
167
- "cacheRead": 0,
192
+ "cacheRead": 0.16,
168
193
  "cacheWrite": 0
169
194
  },
170
195
  "contextWindow": 262000,
@@ -181,12 +206,13 @@
181
206
  "name": "Kimi K3",
182
207
  "reasoning": true,
183
208
  "input": [
184
- "text"
209
+ "text",
210
+ "image"
185
211
  ],
186
212
  "cost": {
187
213
  "input": 3,
188
214
  "output": 15,
189
- "cacheRead": 0,
215
+ "cacheRead": 0.3,
190
216
  "cacheWrite": 0
191
217
  },
192
218
  "contextWindow": 1048576,
@@ -198,28 +224,6 @@
198
224
  "thinkingFormat": "openai"
199
225
  }
200
226
  },
201
- {
202
- "id": "inception/mercury-2",
203
- "name": "Mercury 2",
204
- "reasoning": true,
205
- "input": [
206
- "text"
207
- ],
208
- "cost": {
209
- "input": 0,
210
- "output": 0,
211
- "cacheRead": 0,
212
- "cacheWrite": 0
213
- },
214
- "contextWindow": 8192,
215
- "maxTokens": 5000,
216
- "compat": {
217
- "supportsDeveloperRole": true,
218
- "supportsStore": false,
219
- "maxTokensField": "max_completion_tokens",
220
- "thinkingFormat": "openai"
221
- }
222
- },
223
227
  {
224
228
  "id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B",
225
229
  "name": "Nemotron Ultra",
@@ -230,7 +234,7 @@
230
234
  "cost": {
231
235
  "input": 0.6,
232
236
  "output": 2.4,
233
- "cacheRead": 0,
237
+ "cacheRead": 0.12,
234
238
  "cacheWrite": 0
235
239
  },
236
240
  "contextWindow": 202800,
@@ -252,7 +256,7 @@
252
256
  "cost": {
253
257
  "input": 0.1,
254
258
  "output": 0.5,
255
- "cacheRead": 0,
259
+ "cacheRead": 0.1,
256
260
  "cacheWrite": 0
257
261
  },
258
262
  "contextWindow": 128072,
@@ -264,27 +268,5 @@
264
268
  "thinkingFormat": "openai",
265
269
  "supportsReasoningEffort": true
266
270
  }
267
- },
268
- {
269
- "id": "sid/sid-1",
270
- "name": "SID-1",
271
- "reasoning": true,
272
- "input": [
273
- "text"
274
- ],
275
- "cost": {
276
- "input": 0,
277
- "output": 0,
278
- "cacheRead": 0,
279
- "cacheWrite": 0
280
- },
281
- "contextWindow": 32768,
282
- "maxTokens": 5000,
283
- "compat": {
284
- "supportsDeveloperRole": true,
285
- "supportsStore": false,
286
- "maxTokensField": "max_completion_tokens",
287
- "thinkingFormat": "openai"
288
- }
289
271
  }
290
272
  ]
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-baseten-provider",
3
- "version": "1.0.8",
3
+ "version": "1.0.10",
4
4
  "description": "Baseten provider extension for pi - Access DeepSeek, Kimi, GLM, MiniMax, Nemotron, and GPT-OSS models through the Baseten Model API",
5
5
  "author": "monotykamary",
6
6
  "homepage": "https://github.com/monotykamary/pi-baseten-provider#readme",
package/patch.json CHANGED
@@ -1,54 +1,220 @@
1
1
  {
2
- "moonshotai/Kimi-K2.6": {
3
- "reasoning": true,
4
- "cost": {
5
- "input": 0.6,
6
- "output": 3.0
2
+ "deepseek-ai/DeepSeek-V4-Flash-0731": {
3
+ "maxTokens": 384000,
4
+ "thinkingLevelMap": {
5
+ "off": "none",
6
+ "minimal": null,
7
+ "low": "low",
8
+ "medium": null,
9
+ "high": "high",
10
+ "xhigh": null,
11
+ "max": "max"
7
12
  },
8
13
  "compat": {
9
- "thinkingFormat": "qwen-chat-template"
14
+ "thinkingFormat": "chat-template",
15
+ "supportsReasoningEffort": false,
16
+ "requiresReasoningContentOnAssistantMessages": true,
17
+ "chatTemplateKwargs": {
18
+ "thinking": {
19
+ "$var": "thinking.enabled"
20
+ },
21
+ "reasoning_effort": {
22
+ "$var": "thinking.effort"
23
+ }
24
+ }
10
25
  }
11
26
  },
12
- "moonshotai/Kimi-K2.5": {
13
- "reasoning": true,
27
+ "deepseek-ai/DeepSeek-V4-Pro": {
28
+ "thinkingLevelMap": {
29
+ "off": "none",
30
+ "minimal": "minimal",
31
+ "low": "low",
32
+ "medium": "medium",
33
+ "high": "high",
34
+ "xhigh": "xhigh",
35
+ "max": "max"
36
+ },
14
37
  "compat": {
15
- "thinkingFormat": "qwen-chat-template"
38
+ "thinkingFormat": "openai",
39
+ "supportsReasoningEffort": true,
40
+ "requiresReasoningContentOnAssistantMessages": true
16
41
  }
17
42
  },
18
43
  "zai-org/GLM-4.7": {
19
44
  "reasoning": true,
45
+ "thinkingLevelMap": {
46
+ "off": "none",
47
+ "minimal": null,
48
+ "low": null,
49
+ "medium": null,
50
+ "high": "high",
51
+ "xhigh": null,
52
+ "max": null
53
+ },
20
54
  "compat": {
21
- "thinkingFormat": "qwen-chat-template"
55
+ "thinkingFormat": "qwen-chat-template",
56
+ "supportsReasoningEffort": false,
57
+ "requiresReasoningContentOnAssistantMessages": true
22
58
  }
23
59
  },
24
- "zai-org/GLM-5": {
25
- "reasoning": true,
60
+ "zai-org/GLM-5.2": {
61
+ "thinkingLevelMap": {
62
+ "off": "none",
63
+ "minimal": null,
64
+ "low": null,
65
+ "medium": null,
66
+ "high": "high",
67
+ "xhigh": null,
68
+ "max": "max"
69
+ },
26
70
  "compat": {
27
- "thinkingFormat": "qwen-chat-template"
71
+ "thinkingFormat": "openai",
72
+ "supportsReasoningEffort": true,
73
+ "requiresReasoningContentOnAssistantMessages": true
28
74
  }
29
75
  },
30
- "zai-org/GLM-5.2": {
76
+ "zai-org/GLM-5.2-Fast": {
77
+ "thinkingLevelMap": {
78
+ "off": "none",
79
+ "minimal": null,
80
+ "low": null,
81
+ "medium": null,
82
+ "high": "high",
83
+ "xhigh": null,
84
+ "max": "max"
85
+ },
31
86
  "compat": {
32
- "thinkingFormat": "qwen-chat-template"
87
+ "thinkingFormat": "openai",
88
+ "supportsReasoningEffort": true,
89
+ "requiresReasoningContentOnAssistantMessages": true
90
+ }
91
+ },
92
+ "thinkingmachines/inkling": {
93
+ "thinkingLevelMap": {
94
+ "off": "none",
95
+ "minimal": "minimal",
96
+ "low": "low",
97
+ "medium": "medium",
98
+ "high": "high",
99
+ "xhigh": "xhigh",
100
+ "max": "max"
33
101
  },
102
+ "compat": {
103
+ "thinkingFormat": "openai",
104
+ "supportsReasoningEffort": true
105
+ }
106
+ },
107
+ "thinkingmachines/inkling-small": {
34
108
  "thinkingLevelMap": {
109
+ "off": "none",
35
110
  "minimal": "minimal",
36
111
  "low": "low",
37
112
  "medium": "medium",
38
113
  "high": "high",
114
+ "xhigh": "xhigh",
39
115
  "max": "max"
116
+ },
117
+ "compat": {
118
+ "thinkingFormat": "openai",
119
+ "supportsReasoningEffort": true
120
+ }
121
+ },
122
+ "moonshotai/Kimi-K2.6": {
123
+ "thinkingLevelMap": {
124
+ "off": "none",
125
+ "minimal": null,
126
+ "low": null,
127
+ "medium": null,
128
+ "high": "high",
129
+ "xhigh": null,
130
+ "max": null
131
+ },
132
+ "compat": {
133
+ "thinkingFormat": "qwen-chat-template",
134
+ "supportsReasoningEffort": false,
135
+ "requiresReasoningContentOnAssistantMessages": true
40
136
  }
41
137
  },
42
138
  "moonshotai/Kimi-K2.7-Code": {
43
139
  "thinkingLevelMap": {
140
+ "off": "none",
141
+ "minimal": null,
142
+ "low": null,
143
+ "medium": null,
144
+ "high": "high",
145
+ "xhigh": null,
146
+ "max": null
147
+ },
148
+ "compat": {
149
+ "thinkingFormat": "qwen-chat-template",
150
+ "supportsReasoningEffort": false,
151
+ "requiresReasoningContentOnAssistantMessages": true
152
+ }
153
+ },
154
+ "moonshotai/Kimi-K3": {
155
+ "thinkingLevelMap": {
156
+ "off": "none",
157
+ "minimal": null,
158
+ "low": "low",
159
+ "medium": null,
160
+ "high": "high",
161
+ "xhigh": null,
162
+ "max": "max"
163
+ },
164
+ "compat": {
165
+ "thinkingFormat": "openai",
166
+ "supportsReasoningEffort": true,
167
+ "requiresReasoningContentOnAssistantMessages": true
168
+ }
169
+ },
170
+ "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B": {
171
+ "thinkingLevelMap": {
172
+ "off": "none",
44
173
  "minimal": null,
174
+ "low": null,
175
+ "medium": null,
176
+ "high": "high",
177
+ "xhigh": null,
178
+ "max": null
179
+ },
180
+ "compat": {
181
+ "thinkingFormat": "qwen-chat-template",
182
+ "supportsReasoningEffort": false,
183
+ "requiresReasoningContentOnAssistantMessages": true
184
+ }
185
+ },
186
+ "openai/gpt-oss-120b": {
187
+ "thinkingLevelMap": {
188
+ "off": "none",
189
+ "minimal": "minimal",
190
+ "low": "low",
191
+ "medium": "medium",
192
+ "high": "high",
193
+ "xhigh": "xhigh",
194
+ "max": "max"
195
+ },
196
+ "compat": {
197
+ "thinkingFormat": "openai",
198
+ "supportsReasoningEffort": true
199
+ }
200
+ },
201
+ "inception/mercury-2": {
202
+ "reasoning": true,
203
+ "thinkingLevelMap": {
204
+ "off": "instant",
205
+ "minimal": "low",
45
206
  "low": "low",
46
207
  "medium": "medium",
47
208
  "high": "high",
48
- "xhigh": null
209
+ "xhigh": null,
210
+ "max": null
211
+ },
212
+ "compat": {
213
+ "thinkingFormat": "openai",
214
+ "supportsReasoningEffort": true
49
215
  }
50
216
  },
51
- "nvidia/Nemotron-120B-A12B": {
52
- "reasoning": true
217
+ "sid/sid-1": {
218
+ "reasoning": false
53
219
  }
54
220
  }
@@ -17,15 +17,119 @@
17
17
  * patch.json and custom-models.json are applied at runtime by the provider.
18
18
  * They are NOT baked into models.json, but ARE used to generate the README table.
19
19
  *
20
- * Requires BASETEN_API_KEY environment variable.
20
+ * API key: the stored `baseten` credential in ~/.pi/agent/auth.json wins, then
21
+ * the BASETEN_API_KEY environment variable. The script refuses to run without one.
21
22
  */
22
23
 
23
24
  import fs from 'fs';
25
+ import os from 'os';
26
+ import { execSync } from 'child_process';
24
27
  import path from 'path';
25
28
  import { fileURLToPath } from 'url';
26
29
 
27
30
  const __dirname = path.dirname(fileURLToPath(import.meta.url));
28
31
 
32
+ // pi's agent directory: PI_CODING_AGENT_DIR (with ~ expansion) or ~/.pi/agent.
33
+ function piAgentDir() {
34
+ const envDir = process.env.PI_CODING_AGENT_DIR;
35
+ if (envDir) {
36
+ return envDir.startsWith('~/') || envDir === '~'
37
+ ? path.join(os.homedir(), envDir.slice(1))
38
+ : envDir;
39
+ }
40
+ return path.join(os.homedir(), '.pi', 'agent');
41
+ }
42
+
43
+ const AUTH_JSON_PATH = path.join(piAgentDir(), 'auth.json');
44
+
45
+ /**
46
+ * Resolve a configured value using pi's semantics (resolve-config-value.ts in
47
+ * pi-mono): "!command" runs via the shell (10s timeout) and uses trimmed
48
+ * stdout; "$VAR" / "${VAR}" interpolate environment variables ("$$" escapes a
49
+ * literal "$", "$!" a literal "!"); anything else is a literal. Returns
50
+ * undefined when a referenced env var is unset or a command fails.
51
+ */
52
+ function resolveConfigValue(config, env) {
53
+ if (typeof config !== 'string' || config.length === 0) return undefined;
54
+ if (config.startsWith('!')) {
55
+ try {
56
+ const out = execSync(config.slice(1), {
57
+ encoding: 'utf8',
58
+ timeout: 10000,
59
+ stdio: ['ignore', 'pipe', 'ignore'],
60
+ });
61
+ return out.trim() || undefined;
62
+ } catch {
63
+ return undefined;
64
+ }
65
+ }
66
+ const ENV_NAME_RE = /^[A-Za-z_][A-Za-z0-9_]*$/;
67
+ let resolved = '';
68
+ let index = 0;
69
+ while (index < config.length) {
70
+ const dollar = config.indexOf('$', index);
71
+ if (dollar < 0) {
72
+ resolved += config.slice(index);
73
+ break;
74
+ }
75
+ resolved += config.slice(index, dollar);
76
+ const next = config[dollar + 1];
77
+ let name;
78
+ if (next === '$' || next === '!') {
79
+ resolved += next;
80
+ index = dollar + 2;
81
+ continue;
82
+ } else if (next === '{') {
83
+ const end = config.indexOf('}', dollar + 2);
84
+ if (end < 0) {
85
+ resolved += '$';
86
+ index = dollar + 1;
87
+ continue;
88
+ }
89
+ const inner = config.slice(dollar + 2, end);
90
+ if (!ENV_NAME_RE.test(inner)) {
91
+ resolved += config.slice(dollar, end + 1);
92
+ index = end + 1;
93
+ continue;
94
+ }
95
+ name = inner;
96
+ index = end + 1;
97
+ } else {
98
+ const match = config.slice(dollar + 1).match(/^[A-Za-z_][A-Za-z0-9_]*/);
99
+ if (!match) {
100
+ resolved += '$';
101
+ index = dollar + 1;
102
+ continue;
103
+ }
104
+ name = match[0];
105
+ index = dollar + 1 + name.length;
106
+ }
107
+ const value = (env && env[name]) || process.env[name] || undefined;
108
+ if (value === undefined) return undefined;
109
+ resolved += value;
110
+ }
111
+ return resolved;
112
+ }
113
+
114
+ /**
115
+ * The API key, resolved the way pi itself resolves it for this provider: the
116
+ * stored `baseten` credential in ~/.pi/agent/auth.json wins, then
117
+ * the BASETEN_API_KEY environment variable.
118
+ */
119
+ function resolveApiKey() {
120
+ try {
121
+ const auth = JSON.parse(fs.readFileSync(AUTH_JSON_PATH, 'utf8'));
122
+ const credential = auth?.baseten;
123
+ if (credential && credential.type === 'api_key' && typeof credential.key === 'string') {
124
+ const key = resolveConfigValue(credential.key, credential.env);
125
+ if (key) return key;
126
+ }
127
+ } catch {
128
+ // Missing or unparseable auth.json: fall through to the env var.
129
+ }
130
+ return process.env.BASETEN_API_KEY || undefined;
131
+ }
132
+
29
133
  const MODELS_API_URL = 'https://inference.baseten.co/v1/models';
30
134
  const MODELS_JSON_PATH = path.join(__dirname, '..', 'models.json');
31
135
  const PATCH_JSON_PATH = path.join(__dirname, '..', 'patch.json');
@@ -58,9 +162,9 @@ function toPerMillion(val) {
58
162
  // ─── API fetch ───────────────────────────────────────────────────────────────
59
163
 
60
164
  async function fetchModels() {
61
- const apiKey = process.env.BASETEN_API_KEY;
165
+ const apiKey = resolveApiKey();
62
166
  if (!apiKey) {
63
- throw new Error('BASETEN_API_KEY environment variable is required');
167
+ throw new Error('No API key found: no `baseten` credential resolved from ' + AUTH_JSON_PATH + ' and BASETEN_API_KEY is not set');
64
168
  }
65
169
 
66
170
  console.log(`Fetching models from ${MODELS_API_URL}...`);
@@ -96,16 +200,17 @@ function transformApiModel(apiModel, existingModelsMap) {
96
200
  }
97
201
  // Update features from API
98
202
  const features = apiModel.supported_features || [];
99
- existing.reasoning = features.includes('reasoning') ?? existing.reasoning;
100
- if (features.includes('vision') && !existing.input.includes('image')) {
101
- existing.input = ['text', 'image'];
102
- }
203
+ existing.reasoning = features.includes('reasoning');
204
+ const hasVision = (apiModel.input_modalities || []).includes('image');
205
+ existing.input = hasVision ? ['text', 'image'] : ['text'];
103
206
  // Update pricing from API
104
207
  const pricing = apiModel.pricing || {};
105
208
  const inputCost = toPerMillion(pricing.prompt);
106
209
  const outputCost = toPerMillion(pricing.completion);
107
- if (inputCost !== null && inputCost > 0) existing.cost.input = inputCost;
108
- if (outputCost !== null && outputCost > 0) existing.cost.output = outputCost;
210
+ const cacheReadCost = toPerMillion(pricing.input_cache_read ?? pricing.cache_prompt);
211
+ if (inputCost !== null) existing.cost.input = inputCost;
212
+ if (outputCost !== null) existing.cost.output = outputCost;
213
+ if (cacheReadCost !== null) existing.cost.cacheRead = cacheReadCost;
109
214
  return existing;
110
215
  }
111
216
 
@@ -113,7 +218,7 @@ function transformApiModel(apiModel, existingModelsMap) {
113
218
  const features = apiModel.supported_features || [];
114
219
  const pricing = apiModel.pricing || {};
115
220
  const hasReasoning = features.includes('reasoning');
116
- const hasVision = features.includes('vision');
221
+ const hasVision = (apiModel.input_modalities || []).includes('image');
117
222
 
118
223
  const inputTypes = ['text'];
119
224
  if (hasVision) inputTypes.push('image');
@@ -130,7 +235,7 @@ function transformApiModel(apiModel, existingModelsMap) {
130
235
  cost: {
131
236
  input: inputCost,
132
237
  output: outputCost,
133
- cacheRead: 0,
238
+ cacheRead: toPerMillion(pricing.input_cache_read ?? pricing.cache_prompt) || 0,
134
239
  cacheWrite: 0,
135
240
  },
136
241
  contextWindow: apiModel.context_length || 131072,