mohdel 0.120.0 → 0.121.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/config/curated.schema.json +408 -75
- package/js/session/adapters/_chat_completions.js +0 -3
- package/js/session/adapters/_speed.js +28 -60
- package/js/session/adapters/anthropic.js +0 -2
- package/js/session/adapters/gemini.js +0 -2
- package/js/session/adapters/openai.js +69 -3
- package/js/session/run.js +19 -13
- package/package.json +5 -5
- package/src/cli/ask.js +6 -0
- package/src/cli/check.js +6 -4
- package/src/lib/index.js +16 -2
- package/src/lib/schema.js +4 -4
package/README.md
CHANGED
|
@@ -102,7 +102,7 @@ mo ask anthropic/claude-sonnet-4-6 --stream "write a haiku about recursion"
|
|
|
102
102
|
mo ask anthropic/claude-opus-4-6 --effort high "prove P != NP"
|
|
103
103
|
|
|
104
104
|
# On a faster service lane, when the model sells one
|
|
105
|
-
mo ask
|
|
105
|
+
mo ask openai/gpt-5.6-luna@fast "triage this alert"
|
|
106
106
|
|
|
107
107
|
# Speech → text from an audio file
|
|
108
108
|
mo transcribe groq/whisper-large-v3-turbo meeting.mp3
|
|
@@ -318,7 +318,7 @@ What each provider supports through mohdel's unified interface:
|
|
|
318
318
|
|
|
319
319
|
Adapter capability ≠ model capability — whether a given model accepts images, tools, or thinking effort depends on the model spec in `curated.json`. The adapter passes through what the envelope supplies; the provider rejects unsupported combos.
|
|
320
320
|
|
|
321
|
-
**Service speeds** are the exception to that pass-through rule. Where a provider sells the same weights at several speeds (
|
|
321
|
+
**Service speeds** are the exception to that pass-through rule. Where a provider sells the same weights at several speeds (OpenAI's `service_tier`, and equivalents elsewhere), the lanes a model sells are declared in its `curated.json` entry under `speeds`, with their own prices and rate limits. Select one with `speed` or the `@lane` id suffix. There is no default lane and no fallback: an undeclared lane fails the call before it is sent, because a model that silently ignores an unsupported lane would otherwise be billed at the lane's rates for standard service. See [docs/CATALOG.md](docs/CATALOG.md#service-speeds).
|
|
322
322
|
|
|
323
323
|
## Local Development
|
|
324
324
|
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
3
|
"$id": "https://github.com/clbrge/mohdel/blob/main/config/curated.schema.json",
|
|
4
4
|
"title": "Mohdel curated.json",
|
|
5
|
-
"description": "Schema for ~/.config/mohdel/curated.json
|
|
5
|
+
"description": "Schema for ~/.config/mohdel/curated.json \u2014 see docs/CATALOG.md for the full reference.",
|
|
6
6
|
"type": "object",
|
|
7
7
|
"propertyNames": {
|
|
8
8
|
"pattern": "^([a-z0-9_-]+/.+|_[a-zA-Z0-9_-]+)$",
|
|
@@ -10,16 +10,25 @@
|
|
|
10
10
|
},
|
|
11
11
|
"additionalProperties": {
|
|
12
12
|
"oneOf": [
|
|
13
|
-
{
|
|
14
|
-
|
|
15
|
-
|
|
13
|
+
{
|
|
14
|
+
"$ref": "#/$defs/deprecatedStub"
|
|
15
|
+
},
|
|
16
|
+
{
|
|
17
|
+
"$ref": "#/$defs/modelEntry"
|
|
18
|
+
},
|
|
19
|
+
{
|
|
20
|
+
"type": "string",
|
|
21
|
+
"description": "Inline comment value for underscore-prefixed keys."
|
|
22
|
+
}
|
|
16
23
|
]
|
|
17
24
|
},
|
|
18
25
|
"$defs": {
|
|
19
26
|
"deprecatedStub": {
|
|
20
27
|
"type": "object",
|
|
21
28
|
"description": "Redirect from a retired model id to its replacement.",
|
|
22
|
-
"required": [
|
|
29
|
+
"required": [
|
|
30
|
+
"deprecated"
|
|
31
|
+
],
|
|
23
32
|
"additionalProperties": true,
|
|
24
33
|
"properties": {
|
|
25
34
|
"deprecated": {
|
|
@@ -31,7 +40,11 @@
|
|
|
31
40
|
"modelEntry": {
|
|
32
41
|
"type": "object",
|
|
33
42
|
"description": "A real model entry. 'model', 'creator', and 'inputFormat' are required.",
|
|
34
|
-
"required": [
|
|
43
|
+
"required": [
|
|
44
|
+
"model",
|
|
45
|
+
"creator",
|
|
46
|
+
"inputFormat"
|
|
47
|
+
],
|
|
35
48
|
"additionalProperties": true,
|
|
36
49
|
"properties": {
|
|
37
50
|
"model": {
|
|
@@ -45,7 +58,15 @@
|
|
|
45
58
|
"inputFormat": {
|
|
46
59
|
"type": "array",
|
|
47
60
|
"description": "Accepted input modalities. Defaults to ['text'].",
|
|
48
|
-
"items": {
|
|
61
|
+
"items": {
|
|
62
|
+
"type": "string",
|
|
63
|
+
"enum": [
|
|
64
|
+
"text",
|
|
65
|
+
"image",
|
|
66
|
+
"video",
|
|
67
|
+
"audio"
|
|
68
|
+
]
|
|
69
|
+
},
|
|
49
70
|
"minItems": 1,
|
|
50
71
|
"uniqueItems": true
|
|
51
72
|
},
|
|
@@ -53,142 +74,454 @@
|
|
|
53
74
|
"type": "string",
|
|
54
75
|
"description": "Routing provider. Defaults to the provider segment of the catalog key."
|
|
55
76
|
},
|
|
56
|
-
"sdk": {
|
|
77
|
+
"sdk": {
|
|
78
|
+
"type": "string",
|
|
79
|
+
"description": "SDK adapter to use (some providers require this)."
|
|
80
|
+
},
|
|
57
81
|
"type": {
|
|
58
82
|
"type": "string",
|
|
59
|
-
"enum": [
|
|
83
|
+
"enum": [
|
|
84
|
+
"model",
|
|
85
|
+
"image",
|
|
86
|
+
"transcription"
|
|
87
|
+
],
|
|
60
88
|
"default": "model",
|
|
61
89
|
"description": "'model' for chat/completion, 'image' for image generation, 'transcription' for speech-to-text."
|
|
62
90
|
},
|
|
63
|
-
"label": {
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
"
|
|
68
|
-
|
|
91
|
+
"label": {
|
|
92
|
+
"type": "string",
|
|
93
|
+
"description": "Human-readable name shown in UIs."
|
|
94
|
+
},
|
|
95
|
+
"description": {
|
|
96
|
+
"type": "string"
|
|
97
|
+
},
|
|
98
|
+
"version": {
|
|
99
|
+
"type": "string"
|
|
100
|
+
},
|
|
101
|
+
"createdAt": {
|
|
102
|
+
"type": "string",
|
|
103
|
+
"format": "date-time"
|
|
104
|
+
},
|
|
105
|
+
"created": {
|
|
106
|
+
"type": "number",
|
|
107
|
+
"description": "Unix timestamp (seconds)."
|
|
108
|
+
},
|
|
69
109
|
"inputPrice": {
|
|
70
110
|
"oneOf": [
|
|
71
|
-
{
|
|
72
|
-
|
|
111
|
+
{
|
|
112
|
+
"type": "number",
|
|
113
|
+
"minimum": 0
|
|
114
|
+
},
|
|
115
|
+
{
|
|
116
|
+
"type": "object",
|
|
117
|
+
"required": [
|
|
118
|
+
"default"
|
|
119
|
+
],
|
|
120
|
+
"additionalProperties": {
|
|
121
|
+
"type": "number"
|
|
122
|
+
},
|
|
123
|
+
"properties": {
|
|
124
|
+
"default": {
|
|
125
|
+
"type": "number"
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
}
|
|
73
129
|
],
|
|
74
|
-
"description": "USD per 1M input tokens. Object form is for tiered pricing
|
|
130
|
+
"description": "USD per 1M input tokens. Object form is for tiered pricing \u2014 must include a 'default' key."
|
|
75
131
|
},
|
|
76
132
|
"outputPrice": {
|
|
77
133
|
"oneOf": [
|
|
78
|
-
{
|
|
79
|
-
|
|
134
|
+
{
|
|
135
|
+
"type": "number",
|
|
136
|
+
"minimum": 0
|
|
137
|
+
},
|
|
138
|
+
{
|
|
139
|
+
"type": "object",
|
|
140
|
+
"required": [
|
|
141
|
+
"default"
|
|
142
|
+
],
|
|
143
|
+
"additionalProperties": {
|
|
144
|
+
"type": "number"
|
|
145
|
+
},
|
|
146
|
+
"properties": {
|
|
147
|
+
"default": {
|
|
148
|
+
"type": "number"
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
}
|
|
80
152
|
],
|
|
81
153
|
"description": "USD per 1M output tokens."
|
|
82
154
|
},
|
|
83
155
|
"thinkingPrice": {
|
|
84
156
|
"oneOf": [
|
|
85
|
-
{
|
|
86
|
-
|
|
157
|
+
{
|
|
158
|
+
"type": "number",
|
|
159
|
+
"minimum": 0
|
|
160
|
+
},
|
|
161
|
+
{
|
|
162
|
+
"type": "object",
|
|
163
|
+
"required": [
|
|
164
|
+
"default"
|
|
165
|
+
],
|
|
166
|
+
"additionalProperties": {
|
|
167
|
+
"type": "number"
|
|
168
|
+
},
|
|
169
|
+
"properties": {
|
|
170
|
+
"default": {
|
|
171
|
+
"type": "number"
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
}
|
|
87
175
|
],
|
|
88
176
|
"description": "USD per 1M thinking/reasoning tokens (when the provider bills these separately)."
|
|
89
177
|
},
|
|
90
|
-
"cacheWritePrice": {
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
178
|
+
"cacheWritePrice": {
|
|
179
|
+
"oneOf": [
|
|
180
|
+
{
|
|
181
|
+
"type": "number",
|
|
182
|
+
"minimum": 0
|
|
183
|
+
},
|
|
184
|
+
{
|
|
185
|
+
"type": "object",
|
|
186
|
+
"required": [
|
|
187
|
+
"default"
|
|
188
|
+
],
|
|
189
|
+
"additionalProperties": {
|
|
190
|
+
"type": "number"
|
|
191
|
+
},
|
|
192
|
+
"properties": {
|
|
193
|
+
"default": {
|
|
194
|
+
"type": "number"
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
],
|
|
199
|
+
"description": "USD per 1M tokens written to provider-side prompt cache. Object form is tiered on input tokens."
|
|
200
|
+
},
|
|
201
|
+
"cacheWrite1hPrice": {
|
|
202
|
+
"oneOf": [
|
|
203
|
+
{
|
|
204
|
+
"type": "number",
|
|
205
|
+
"minimum": 0
|
|
206
|
+
},
|
|
207
|
+
{
|
|
208
|
+
"type": "object",
|
|
209
|
+
"required": [
|
|
210
|
+
"default"
|
|
211
|
+
],
|
|
212
|
+
"additionalProperties": {
|
|
213
|
+
"type": "number"
|
|
214
|
+
},
|
|
215
|
+
"properties": {
|
|
216
|
+
"default": {
|
|
217
|
+
"type": "number"
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
],
|
|
222
|
+
"description": "USD per 1M tokens written with a 1h TTL, when the provider prices that above the default write rate. Falls back to cacheWritePrice."
|
|
223
|
+
},
|
|
224
|
+
"cacheReadPrice": {
|
|
225
|
+
"oneOf": [
|
|
226
|
+
{
|
|
227
|
+
"type": "number",
|
|
228
|
+
"minimum": 0
|
|
229
|
+
},
|
|
230
|
+
{
|
|
231
|
+
"type": "object",
|
|
232
|
+
"required": [
|
|
233
|
+
"default"
|
|
234
|
+
],
|
|
235
|
+
"additionalProperties": {
|
|
236
|
+
"type": "number"
|
|
237
|
+
},
|
|
238
|
+
"properties": {
|
|
239
|
+
"default": {
|
|
240
|
+
"type": "number"
|
|
241
|
+
}
|
|
242
|
+
}
|
|
243
|
+
}
|
|
244
|
+
],
|
|
245
|
+
"description": "USD per 1M tokens served from provider-side prompt cache. Object form is tiered on input tokens."
|
|
246
|
+
},
|
|
247
|
+
"contextTokenLimit": {
|
|
248
|
+
"type": "integer",
|
|
249
|
+
"minimum": 1,
|
|
250
|
+
"description": "Maximum total tokens (input + output)."
|
|
251
|
+
},
|
|
252
|
+
"outputTokenLimit": {
|
|
253
|
+
"type": "integer",
|
|
254
|
+
"minimum": 1,
|
|
255
|
+
"description": "Maximum output tokens per call."
|
|
256
|
+
},
|
|
257
|
+
"thinkingTokenLimit": {
|
|
258
|
+
"type": "integer",
|
|
259
|
+
"minimum": 1,
|
|
260
|
+
"description": "Maximum thinking tokens per call (when separate from output)."
|
|
261
|
+
},
|
|
262
|
+
"tokenizerHeadroom": {
|
|
263
|
+
"type": "number",
|
|
264
|
+
"exclusiveMinimum": 0,
|
|
265
|
+
"description": "Multiplier applied to local token estimates to account for tokenizer drift."
|
|
266
|
+
},
|
|
99
267
|
"thinkingEffortLevels": {
|
|
100
268
|
"oneOf": [
|
|
101
|
-
{
|
|
269
|
+
{
|
|
270
|
+
"type": "null"
|
|
271
|
+
},
|
|
102
272
|
{
|
|
103
273
|
"type": "object",
|
|
104
|
-
"description": "Map effort name
|
|
105
|
-
"additionalProperties": {
|
|
274
|
+
"description": "Map effort name \u2192 provider-native budget. Standard names: low/medium/high/xhigh/max/none.",
|
|
275
|
+
"additionalProperties": {
|
|
276
|
+
"type": "number",
|
|
277
|
+
"minimum": 0
|
|
278
|
+
}
|
|
106
279
|
}
|
|
107
280
|
]
|
|
108
281
|
},
|
|
109
|
-
"defaultThinkingEffort": {
|
|
110
|
-
|
|
282
|
+
"defaultThinkingEffort": {
|
|
283
|
+
"type": "string",
|
|
284
|
+
"description": "Effort level used when the envelope omits 'outputEffort'."
|
|
285
|
+
},
|
|
111
286
|
"speeds": {
|
|
112
287
|
"type": "object",
|
|
113
|
-
"description": "Service speed lanes this model sells, keyed by lane name. A lane is a provider request parameter buying a different speed/price point for the same weights. There is no default lane: omitting 'speed' on the envelope sends no parameter. Only lanes declared here may be requested.",
|
|
288
|
+
"description": "Service speed lanes this model sells, keyed by lane name. A lane is a provider request parameter buying a different speed/price point for the same weights. There is no default lane: omitting 'speed' on the envelope sends no parameter. Only lanes declared here may be requested, and only names the provider's adapter accepts.",
|
|
114
289
|
"additionalProperties": {
|
|
115
290
|
"type": "object",
|
|
116
|
-
"required": ["wire"],
|
|
117
291
|
"additionalProperties": false,
|
|
118
|
-
"description": "Lane overlay
|
|
292
|
+
"description": "Lane overlay: the price and rate-limit fields that differ from the base entry. Unnamed fields fall through. Lane names come from the provider's own vocabulary (OpenAI: fast, priority, flex, scale) \u2014 the adapter decides how a name reaches the wire.",
|
|
119
293
|
"properties": {
|
|
120
|
-
"wire": { "type": "string", "description": "Provider-native value for the lane parameter (e.g. 'fast')." },
|
|
121
294
|
"inputPrice": {
|
|
122
295
|
"oneOf": [
|
|
123
|
-
{
|
|
124
|
-
|
|
296
|
+
{
|
|
297
|
+
"type": "number",
|
|
298
|
+
"minimum": 0
|
|
299
|
+
},
|
|
300
|
+
{
|
|
301
|
+
"type": "object",
|
|
302
|
+
"required": [
|
|
303
|
+
"default"
|
|
304
|
+
],
|
|
305
|
+
"additionalProperties": {
|
|
306
|
+
"type": "number"
|
|
307
|
+
},
|
|
308
|
+
"properties": {
|
|
309
|
+
"default": {
|
|
310
|
+
"type": "number"
|
|
311
|
+
}
|
|
312
|
+
}
|
|
313
|
+
}
|
|
125
314
|
]
|
|
126
315
|
},
|
|
127
316
|
"outputPrice": {
|
|
128
317
|
"oneOf": [
|
|
129
|
-
{
|
|
130
|
-
|
|
318
|
+
{
|
|
319
|
+
"type": "number",
|
|
320
|
+
"minimum": 0
|
|
321
|
+
},
|
|
322
|
+
{
|
|
323
|
+
"type": "object",
|
|
324
|
+
"required": [
|
|
325
|
+
"default"
|
|
326
|
+
],
|
|
327
|
+
"additionalProperties": {
|
|
328
|
+
"type": "number"
|
|
329
|
+
},
|
|
330
|
+
"properties": {
|
|
331
|
+
"default": {
|
|
332
|
+
"type": "number"
|
|
333
|
+
}
|
|
334
|
+
}
|
|
335
|
+
}
|
|
131
336
|
]
|
|
132
337
|
},
|
|
133
338
|
"thinkingPrice": {
|
|
134
339
|
"oneOf": [
|
|
135
|
-
{
|
|
136
|
-
|
|
340
|
+
{
|
|
341
|
+
"type": "number",
|
|
342
|
+
"minimum": 0
|
|
343
|
+
},
|
|
344
|
+
{
|
|
345
|
+
"type": "object",
|
|
346
|
+
"required": [
|
|
347
|
+
"default"
|
|
348
|
+
],
|
|
349
|
+
"additionalProperties": {
|
|
350
|
+
"type": "number"
|
|
351
|
+
},
|
|
352
|
+
"properties": {
|
|
353
|
+
"default": {
|
|
354
|
+
"type": "number"
|
|
355
|
+
}
|
|
356
|
+
}
|
|
357
|
+
}
|
|
358
|
+
]
|
|
359
|
+
},
|
|
360
|
+
"cacheWritePrice": {
|
|
361
|
+
"oneOf": [
|
|
362
|
+
{
|
|
363
|
+
"type": "number",
|
|
364
|
+
"minimum": 0
|
|
365
|
+
},
|
|
366
|
+
{
|
|
367
|
+
"type": "object",
|
|
368
|
+
"required": [
|
|
369
|
+
"default"
|
|
370
|
+
],
|
|
371
|
+
"additionalProperties": {
|
|
372
|
+
"type": "number"
|
|
373
|
+
},
|
|
374
|
+
"properties": {
|
|
375
|
+
"default": {
|
|
376
|
+
"type": "number"
|
|
377
|
+
}
|
|
378
|
+
}
|
|
379
|
+
}
|
|
137
380
|
]
|
|
138
381
|
},
|
|
139
|
-
"
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
382
|
+
"cacheWrite1hPrice": {
|
|
383
|
+
"oneOf": [
|
|
384
|
+
{
|
|
385
|
+
"type": "number",
|
|
386
|
+
"minimum": 0
|
|
387
|
+
},
|
|
388
|
+
{
|
|
389
|
+
"type": "object",
|
|
390
|
+
"required": [
|
|
391
|
+
"default"
|
|
392
|
+
],
|
|
393
|
+
"additionalProperties": {
|
|
394
|
+
"type": "number"
|
|
395
|
+
},
|
|
396
|
+
"properties": {
|
|
397
|
+
"default": {
|
|
398
|
+
"type": "number"
|
|
399
|
+
}
|
|
400
|
+
}
|
|
401
|
+
}
|
|
402
|
+
]
|
|
403
|
+
},
|
|
404
|
+
"cacheReadPrice": {
|
|
405
|
+
"oneOf": [
|
|
406
|
+
{
|
|
407
|
+
"type": "number",
|
|
408
|
+
"minimum": 0
|
|
409
|
+
},
|
|
410
|
+
{
|
|
411
|
+
"type": "object",
|
|
412
|
+
"required": [
|
|
413
|
+
"default"
|
|
414
|
+
],
|
|
415
|
+
"additionalProperties": {
|
|
416
|
+
"type": "number"
|
|
417
|
+
},
|
|
418
|
+
"properties": {
|
|
419
|
+
"default": {
|
|
420
|
+
"type": "number"
|
|
421
|
+
}
|
|
422
|
+
}
|
|
423
|
+
}
|
|
424
|
+
]
|
|
425
|
+
},
|
|
426
|
+
"rpmLimit": {
|
|
427
|
+
"type": "integer",
|
|
428
|
+
"minimum": 1
|
|
429
|
+
},
|
|
430
|
+
"tpmLimit": {
|
|
431
|
+
"type": "integer",
|
|
432
|
+
"minimum": 1
|
|
433
|
+
}
|
|
144
434
|
}
|
|
145
435
|
}
|
|
146
436
|
},
|
|
147
|
-
|
|
148
437
|
"tags": {
|
|
149
438
|
"type": "array",
|
|
150
|
-
"items": {
|
|
439
|
+
"items": {
|
|
440
|
+
"type": "string",
|
|
441
|
+
"pattern": "^[a-zA-Z][a-zA-Z0-9._-]{0,31}$"
|
|
442
|
+
},
|
|
151
443
|
"uniqueItems": true,
|
|
152
444
|
"description": "Free-form classification tags (used by 'mo bench --tag', 'mo rank', and caller-side selection)."
|
|
153
445
|
},
|
|
154
446
|
"aliases": {
|
|
155
447
|
"type": "array",
|
|
156
|
-
"items": {
|
|
448
|
+
"items": {
|
|
449
|
+
"type": "string"
|
|
450
|
+
},
|
|
157
451
|
"description": "Alternative ids that should resolve to this entry."
|
|
158
452
|
},
|
|
159
453
|
"replaces": {
|
|
160
454
|
"type": "array",
|
|
161
|
-
"items": {
|
|
455
|
+
"items": {
|
|
456
|
+
"type": "string"
|
|
457
|
+
},
|
|
162
458
|
"description": "Older model ids this one supersedes."
|
|
163
459
|
},
|
|
164
460
|
"leaderboard": {
|
|
165
461
|
"type": "array",
|
|
166
|
-
"items": {
|
|
462
|
+
"items": {
|
|
463
|
+
"type": "number"
|
|
464
|
+
},
|
|
167
465
|
"minItems": 3,
|
|
168
466
|
"maxItems": 3,
|
|
169
|
-
"description": "[intelligence, speed, latency] triple
|
|
170
|
-
},
|
|
171
|
-
"leaderboardNote": {
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
"
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
467
|
+
"description": "[intelligence, speed, latency] triple \u2014 drives 'mo rank'."
|
|
468
|
+
},
|
|
469
|
+
"leaderboardNote": {
|
|
470
|
+
"type": "string"
|
|
471
|
+
},
|
|
472
|
+
"supportsTools": {
|
|
473
|
+
"type": "boolean",
|
|
474
|
+
"description": "Set false to mark a model as tool-less."
|
|
475
|
+
},
|
|
476
|
+
"imagePrice": {
|
|
477
|
+
"type": "number",
|
|
478
|
+
"minimum": 0,
|
|
479
|
+
"description": "USD per generated image (image-type entries)."
|
|
480
|
+
},
|
|
481
|
+
"imageEndpoint": {
|
|
482
|
+
"type": "string",
|
|
483
|
+
"description": "Provider-side image endpoint name."
|
|
484
|
+
},
|
|
485
|
+
"imageDefaultSize": {
|
|
486
|
+
"type": "string",
|
|
487
|
+
"description": "Default size when envelope omits one (e.g. '1024x1024')."
|
|
488
|
+
},
|
|
489
|
+
"transcriptionPrice": {
|
|
490
|
+
"type": "number",
|
|
491
|
+
"minimum": 0,
|
|
492
|
+
"description": "USD per audio minute (transcription-type entries). Token-billed transcription models (OpenAI gpt-4o-*-transcribe) use inputPrice/outputPrice instead."
|
|
493
|
+
},
|
|
494
|
+
"rpmLimit": {
|
|
495
|
+
"type": "integer",
|
|
496
|
+
"minimum": 1,
|
|
497
|
+
"description": "Requests per minute. Overrides provider default."
|
|
498
|
+
},
|
|
499
|
+
"tpmLimit": {
|
|
500
|
+
"type": "integer",
|
|
501
|
+
"minimum": 1,
|
|
502
|
+
"description": "Tokens per minute. Overrides provider default."
|
|
503
|
+
},
|
|
182
504
|
"rateLimitScope": {
|
|
183
505
|
"type": "string",
|
|
184
|
-
"enum": [
|
|
506
|
+
"enum": [
|
|
507
|
+
"model",
|
|
508
|
+
"provider"
|
|
509
|
+
],
|
|
185
510
|
"description": "'model' = private budget. 'provider' = shared with provider-level pool."
|
|
186
511
|
},
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
"
|
|
512
|
+
"deprecated": {
|
|
513
|
+
"type": "string",
|
|
514
|
+
"description": "If present, this entry is treated as a stub (use 'deprecatedStub' shape)."
|
|
515
|
+
},
|
|
516
|
+
"suspended": {
|
|
517
|
+
"type": "string",
|
|
518
|
+
"description": "Reason this model is temporarily disabled."
|
|
519
|
+
},
|
|
520
|
+
"displayName": {
|
|
521
|
+
"type": "string",
|
|
522
|
+
"deprecated": true,
|
|
523
|
+
"description": "Deprecated \u2014 use 'label'."
|
|
524
|
+
}
|
|
192
525
|
}
|
|
193
526
|
}
|
|
194
527
|
}
|
|
@@ -19,7 +19,6 @@
|
|
|
19
19
|
import { getSpec } from './_catalog.js'
|
|
20
20
|
import { classifyProviderError } from './_errors.js'
|
|
21
21
|
import { costFor } from './_pricing.js'
|
|
22
|
-
import { applySpeed } from './_speed.js'
|
|
23
22
|
import { catalogKey, bareOf } from '#core/model-id.js'
|
|
24
23
|
import {
|
|
25
24
|
STATUS_COMPLETED,
|
|
@@ -371,8 +370,6 @@ function buildRequest (envelope, spec, config) {
|
|
|
371
370
|
args[config.identifierField || 'user'] = envelope.identifier
|
|
372
371
|
}
|
|
373
372
|
|
|
374
|
-
applySpeed(args, envelope, spec)
|
|
375
|
-
|
|
376
373
|
return args
|
|
377
374
|
}
|
|
378
375
|
|
|
@@ -1,30 +1,21 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Service-speed lanes.
|
|
2
|
+
* Service-speed lanes — catalog side.
|
|
3
3
|
*
|
|
4
|
-
* A lane is a
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
4
|
+
* A lane is a named service speed a model sells, declared per catalog
|
|
5
|
+
* entry under `speeds` with the price and rate-limit fields that
|
|
6
|
+
* differ from the base entry. This module owns only what is common to
|
|
7
|
+
* every provider: what the entry declares, and what that means for
|
|
8
|
+
* pricing and throttling.
|
|
9
9
|
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
10
|
+
* How a lane reaches the wire, and how the served lane is read back,
|
|
11
|
+
* is provider protocol and lives in the adapter. An adapter that
|
|
12
|
+
* handles lanes advertises the names it accepts on `speedLanes`;
|
|
13
|
+
* absence of that property is the declaration that it handles none,
|
|
14
|
+
* and is what `run.js` checks before dispatch.
|
|
13
15
|
*
|
|
14
16
|
* @module session/adapters/_speed
|
|
15
17
|
*/
|
|
16
18
|
|
|
17
|
-
import { providerOf } from '#core/model-id.js'
|
|
18
|
-
|
|
19
|
-
/**
|
|
20
|
-
* Providers whose adapter emits a lane parameter, mapped to the
|
|
21
|
-
* native parameter name. Absence from this table is the declaration
|
|
22
|
-
* that a provider has no lanes.
|
|
23
|
-
*/
|
|
24
|
-
export const SPEED_PARAMS = Object.freeze({
|
|
25
|
-
anthropic: 'speed'
|
|
26
|
-
})
|
|
27
|
-
|
|
28
19
|
/** Spec fields a lane overlay may restate. */
|
|
29
20
|
export const SPEED_OVERRIDABLE = Object.freeze([
|
|
30
21
|
'inputPrice',
|
|
@@ -37,25 +28,9 @@ export const SPEED_OVERRIDABLE = Object.freeze([
|
|
|
37
28
|
'tpmLimit'
|
|
38
29
|
])
|
|
39
30
|
|
|
40
|
-
/**
|
|
41
|
-
* @param {string} provider
|
|
42
|
-
* @returns {boolean}
|
|
43
|
-
*/
|
|
44
|
-
export function providerSupportsSpeed (provider) {
|
|
45
|
-
return Object.hasOwn(SPEED_PARAMS, provider)
|
|
46
|
-
}
|
|
47
|
-
|
|
48
|
-
/**
|
|
49
|
-
* @param {string} provider
|
|
50
|
-
* @returns {string | undefined}
|
|
51
|
-
*/
|
|
52
|
-
export function speedParamFor (provider) {
|
|
53
|
-
return SPEED_PARAMS[provider]
|
|
54
|
-
}
|
|
55
|
-
|
|
56
31
|
/**
|
|
57
32
|
* Whether `spec` declares `speed`. Uses `hasOwn` rather than
|
|
58
|
-
* truthiness so
|
|
33
|
+
* truthiness so a lane sold at base prices still counts.
|
|
59
34
|
*
|
|
60
35
|
* @param {any} spec
|
|
61
36
|
* @param {string} speed
|
|
@@ -73,6 +48,22 @@ export function speedNames (spec) {
|
|
|
73
48
|
return spec?.speeds ? Object.keys(spec.speeds) : []
|
|
74
49
|
}
|
|
75
50
|
|
|
51
|
+
/**
|
|
52
|
+
* Whether the lane declares a quota of its own, which is what earns it
|
|
53
|
+
* a private rate-limit bucket. Lanes that don't (OpenAI's service_tier
|
|
54
|
+
* shares the model's TPM/RPM pool) count against the base bucket —
|
|
55
|
+
* giving them their own would silently double the allowance.
|
|
56
|
+
*
|
|
57
|
+
* @param {any} spec
|
|
58
|
+
* @param {string} [speed]
|
|
59
|
+
* @returns {boolean}
|
|
60
|
+
*/
|
|
61
|
+
export function speedHasOwnQuota (spec, speed) {
|
|
62
|
+
if (!speed || !hasSpeed(spec, speed)) return false
|
|
63
|
+
const overlay = spec.speeds[speed]
|
|
64
|
+
return overlay.rpmLimit != null || overlay.tpmLimit != null
|
|
65
|
+
}
|
|
66
|
+
|
|
76
67
|
/**
|
|
77
68
|
* Spec with `speed`'s overlay applied. Unnamed fields fall through to
|
|
78
69
|
* the base entry.
|
|
@@ -94,26 +85,3 @@ export function mergeSpeed (spec, speed) {
|
|
|
94
85
|
}
|
|
95
86
|
return merged
|
|
96
87
|
}
|
|
97
|
-
|
|
98
|
-
/**
|
|
99
|
-
* Set the provider-native lane parameter on an outbound request.
|
|
100
|
-
* No-op when the envelope carries no lane.
|
|
101
|
-
*
|
|
102
|
-
* @param {Record<string, any>} request
|
|
103
|
-
* @param {import('#core/envelope.js').CallEnvelope} envelope
|
|
104
|
-
* @param {any} spec
|
|
105
|
-
* @throws when the envelope carries a lane the provider cannot emit
|
|
106
|
-
* or the spec does not declare
|
|
107
|
-
*/
|
|
108
|
-
export function applySpeed (request, envelope, spec) {
|
|
109
|
-
if (!envelope.speed) return
|
|
110
|
-
const provider = providerOf(envelope.model)
|
|
111
|
-
const param = speedParamFor(provider)
|
|
112
|
-
if (!param) {
|
|
113
|
-
throw new Error(`provider '${provider}' does not implement speed lanes`)
|
|
114
|
-
}
|
|
115
|
-
if (!hasSpeed(spec, envelope.speed)) {
|
|
116
|
-
throw new Error(`model does not declare speed lane '${envelope.speed}'`)
|
|
117
|
-
}
|
|
118
|
-
request[param] = spec.speeds[envelope.speed].wire
|
|
119
|
-
}
|
|
@@ -29,7 +29,6 @@ import { classifyProviderError } from './_errors.js'
|
|
|
29
29
|
import { loadImages } from './_images.js'
|
|
30
30
|
import { isTrustedMedia } from './_media.js'
|
|
31
31
|
import { costFor } from './_pricing.js'
|
|
32
|
-
import { applySpeed } from './_speed.js'
|
|
33
32
|
import { catalogKey, bareOf } from '#core/model-id.js'
|
|
34
33
|
import {
|
|
35
34
|
toAnthropicTools,
|
|
@@ -353,7 +352,6 @@ function buildRequest (envelope, conversation, system, conversationCacheTtl = nu
|
|
|
353
352
|
}
|
|
354
353
|
}
|
|
355
354
|
|
|
356
|
-
applySpeed(request, envelope, spec)
|
|
357
355
|
applyCacheBreakpoints(request, conversationCacheTtl)
|
|
358
356
|
|
|
359
357
|
return request
|
|
@@ -31,7 +31,6 @@ import { loadImages } from './_images.js'
|
|
|
31
31
|
import { isTrustedMedia } from './_media.js'
|
|
32
32
|
import { loadVideos } from './_videos.js'
|
|
33
33
|
import { costFor } from './_pricing.js'
|
|
34
|
-
import { applySpeed } from './_speed.js'
|
|
35
34
|
import { catalogKey, bareOf } from '#core/model-id.js'
|
|
36
35
|
import {
|
|
37
36
|
toGeminiTools,
|
|
@@ -272,7 +271,6 @@ function buildRequest (envelope, contents, systemInstruction) {
|
|
|
272
271
|
contents
|
|
273
272
|
}
|
|
274
273
|
if (Object.keys(config).length > 0) request.config = config
|
|
275
|
-
applySpeed(request, envelope, spec)
|
|
276
274
|
return request
|
|
277
275
|
}
|
|
278
276
|
|
|
@@ -30,7 +30,6 @@ import { classifyProviderError } from './_errors.js'
|
|
|
30
30
|
import { loadImages } from './_images.js'
|
|
31
31
|
import { isTrustedMedia } from './_media.js'
|
|
32
32
|
import { costFor } from './_pricing.js'
|
|
33
|
-
import { applySpeed } from './_speed.js'
|
|
34
33
|
import { catalogKey, providerOf, bareOf } from '#core/model-id.js'
|
|
35
34
|
import {
|
|
36
35
|
toOpenAITools,
|
|
@@ -84,6 +83,8 @@ export async function * openai (envelope, deps = {}) {
|
|
|
84
83
|
let status = STATUS_COMPLETED
|
|
85
84
|
/** @type {string | undefined} */
|
|
86
85
|
let warning
|
|
86
|
+
/** @type {string | null | undefined} */
|
|
87
|
+
let servedTier
|
|
87
88
|
|
|
88
89
|
// Tool-call accumulation: itemId → {call_id, name, arguments}
|
|
89
90
|
/** @type {Map<string, {call_id: string, name: string, arguments: string}>} */
|
|
@@ -131,6 +132,7 @@ export async function * openai (envelope, deps = {}) {
|
|
|
131
132
|
break
|
|
132
133
|
|
|
133
134
|
case 'response.completed':
|
|
135
|
+
servedTier = event.response?.service_tier ?? servedTier
|
|
134
136
|
if (event.response?.usage) {
|
|
135
137
|
inputTokens = event.response.usage.input_tokens ?? 0
|
|
136
138
|
outputTokens = event.response.usage.output_tokens ?? 0
|
|
@@ -144,6 +146,7 @@ export async function * openai (envelope, deps = {}) {
|
|
|
144
146
|
break
|
|
145
147
|
|
|
146
148
|
case 'response.incomplete':
|
|
149
|
+
servedTier = event.response?.service_tier ?? servedTier
|
|
147
150
|
status = STATUS_INCOMPLETE
|
|
148
151
|
if (event.response?.incomplete_details?.reason === 'max_output_tokens') {
|
|
149
152
|
warning = WARNING_INSUFFICIENT_OUTPUT_BUDGET
|
|
@@ -190,6 +193,8 @@ export async function * openai (envelope, deps = {}) {
|
|
|
190
193
|
// simpler with the additive shape.
|
|
191
194
|
const regularInputTokens = Math.max(0, inputTokens - cachedInputTokens - cacheWriteTokens)
|
|
192
195
|
|
|
196
|
+
const billed = billedEnvelope(envelope, servedTier, log)
|
|
197
|
+
|
|
193
198
|
/** @type {import('#core/events.js').DoneEvent} */
|
|
194
199
|
const done = {
|
|
195
200
|
type: 'done',
|
|
@@ -201,8 +206,9 @@ export async function * openai (envelope, deps = {}) {
|
|
|
201
206
|
thinkingTokens,
|
|
202
207
|
...(cacheWriteTokens > 0 && { cacheWriteInputTokens: cacheWriteTokens }),
|
|
203
208
|
...(cachedInputTokens > 0 && { cacheReadInputTokens: cachedInputTokens }),
|
|
209
|
+
...(envelope.speed && { servedSpeed: billed.served }),
|
|
204
210
|
cost: costFor(
|
|
205
|
-
envelope,
|
|
211
|
+
billed.envelope,
|
|
206
212
|
{
|
|
207
213
|
inputTokens: regularInputTokens,
|
|
208
214
|
outputTokens: messageOutputTokens,
|
|
@@ -221,6 +227,66 @@ export async function * openai (envelope, deps = {}) {
|
|
|
221
227
|
yield done
|
|
222
228
|
}
|
|
223
229
|
|
|
230
|
+
/**
|
|
231
|
+
* Lane names the OpenAI adapter can put on `service_tier`. `run.js`
|
|
232
|
+
* reads this before dispatch; a lane outside it never reaches here.
|
|
233
|
+
*/
|
|
234
|
+
openai.speedLanes = new Set(['fast', 'priority', 'flex', 'scale'])
|
|
235
|
+
|
|
236
|
+
/**
|
|
237
|
+
* Lane the response says was served, given the one requested.
|
|
238
|
+
*
|
|
239
|
+
* OpenAI answers `service_tier: 'priority'` to a granted request for
|
|
240
|
+
* either premium lane, so the echo alone cannot name which was asked
|
|
241
|
+
* for — the request is the other half of the answer. Any other value
|
|
242
|
+
* is the tier that actually ran, `null` when nothing was reported.
|
|
243
|
+
*
|
|
244
|
+
* @param {string} requested
|
|
245
|
+
* @param {string | null | undefined} servedTier
|
|
246
|
+
* @returns {string | null | undefined}
|
|
247
|
+
*/
|
|
248
|
+
function servedLane (requested, servedTier) {
|
|
249
|
+
if (servedTier == null) return undefined
|
|
250
|
+
if (servedTier === 'priority') {
|
|
251
|
+
return (requested === 'fast' || requested === 'priority') ? requested : 'priority'
|
|
252
|
+
}
|
|
253
|
+
return openai.speedLanes.has(servedTier) ? servedTier : null
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
/**
|
|
257
|
+
* Envelope to price the call against. OpenAI serves Standard instead
|
|
258
|
+
* when a premium lane is unavailable — documented behaviour above the
|
|
259
|
+
* ramp rate limit — so billing the requested lane would charge premium
|
|
260
|
+
* rates for standard service.
|
|
261
|
+
*
|
|
262
|
+
* When a lane was requested and nothing was reported back, the request
|
|
263
|
+
* is the only evidence available: bill it and say so.
|
|
264
|
+
*
|
|
265
|
+
* @param {import('#core/envelope.js').CallEnvelope} envelope
|
|
266
|
+
* @param {string | null | undefined} servedTier
|
|
267
|
+
* @param {any} [log]
|
|
268
|
+
* @returns {{envelope: import('#core/envelope.js').CallEnvelope, served: string | null}}
|
|
269
|
+
*/
|
|
270
|
+
function billedEnvelope (envelope, servedTier, log) {
|
|
271
|
+
if (!envelope.speed) return { envelope, served: null }
|
|
272
|
+
|
|
273
|
+
const served = servedLane(envelope.speed, servedTier)
|
|
274
|
+
if (served === undefined) {
|
|
275
|
+
log?.warn(
|
|
276
|
+
{ model: envelope.model, speed: envelope.speed },
|
|
277
|
+
'[mohdel:openai] no service_tier reported; billing the requested lane'
|
|
278
|
+
)
|
|
279
|
+
return { envelope, served: envelope.speed }
|
|
280
|
+
}
|
|
281
|
+
if (served === envelope.speed) return { envelope, served }
|
|
282
|
+
|
|
283
|
+
log?.warn(
|
|
284
|
+
{ model: envelope.model, requested: envelope.speed, servedTier, served },
|
|
285
|
+
'[mohdel:openai] provider served a different speed lane; billing what was served'
|
|
286
|
+
)
|
|
287
|
+
return { envelope: { ...envelope, speed: served ?? undefined }, served }
|
|
288
|
+
}
|
|
289
|
+
|
|
224
290
|
/**
|
|
225
291
|
* @param {import('#core/envelope.js').CallEnvelope} envelope
|
|
226
292
|
* @param {Array<any>} input
|
|
@@ -288,7 +354,7 @@ function buildRequest (envelope, input, instructions) {
|
|
|
288
354
|
}
|
|
289
355
|
}
|
|
290
356
|
|
|
291
|
-
|
|
357
|
+
if (envelope.speed) request.service_tier = envelope.speed
|
|
292
358
|
|
|
293
359
|
return request
|
|
294
360
|
}
|
package/js/session/run.js
CHANGED
|
@@ -26,7 +26,7 @@ import { getAdapter } from './adapters/index.js'
|
|
|
26
26
|
import { isImageProvider } from './adapters/image/index.js'
|
|
27
27
|
import { getSpec } from './adapters/_catalog.js'
|
|
28
28
|
import { getProviderLimits } from './adapters/_providers.js'
|
|
29
|
-
import { hasSpeed, mergeSpeed,
|
|
29
|
+
import { hasSpeed, mergeSpeed, speedHasOwnQuota, speedNames } from './adapters/_speed.js'
|
|
30
30
|
import { providerOf, catalogKey, effortOf, speedOf } from '#core/model-id.js'
|
|
31
31
|
import * as defaultCooldown from './_cooldown.js'
|
|
32
32
|
import * as defaultLimiter from './_rate_limiter.js'
|
|
@@ -128,7 +128,7 @@ export async function * run (envelope, {
|
|
|
128
128
|
}
|
|
129
129
|
|
|
130
130
|
if (envelope.speed) {
|
|
131
|
-
const speedErr = speedError(key, envelope.speed, spec, provider)
|
|
131
|
+
const speedErr = speedError(key, envelope.speed, spec, provider, adapter)
|
|
132
132
|
if (speedErr) {
|
|
133
133
|
log.warn({ provider, speed: envelope.speed }, '[mohdel:answer] unusable speed lane')
|
|
134
134
|
endSpanError(span, new Error(speedErr.error.message))
|
|
@@ -150,12 +150,10 @@ export async function * run (envelope, {
|
|
|
150
150
|
const providerCfg = resolveProviderLimits(provider) || {}
|
|
151
151
|
const rpmLimit = effective?.rpmLimit ?? providerCfg.rpmLimit
|
|
152
152
|
const tpmLimit = effective?.tpmLimit ?? providerCfg.tpmLimit
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
// traffic throttle the lane being paid for.
|
|
156
|
-
const bucketKey = envelope.speed
|
|
153
|
+
const baseBucket = (spec?.rateLimitScope === 'model') ? key : provider
|
|
154
|
+
const bucketKey = speedHasOwnQuota(spec, envelope.speed)
|
|
157
155
|
? `${key}@${envelope.speed}`
|
|
158
|
-
:
|
|
156
|
+
: baseBucket
|
|
159
157
|
|
|
160
158
|
// `0` is a killswitch ("deny all"), not "unset"; `undefined`/`null`
|
|
161
159
|
// means no limit configured for that dimension. Gate on nullability
|
|
@@ -363,25 +361,31 @@ function normalizeModelId (envelope, resolveSpec) {
|
|
|
363
361
|
* @param {string} speed
|
|
364
362
|
* @param {any} spec
|
|
365
363
|
* @param {string} provider
|
|
364
|
+
* @param {any} adapter
|
|
366
365
|
* @returns {import('#core/events.js').ErrorEvent | undefined}
|
|
367
366
|
*/
|
|
368
|
-
function speedError (key, speed, spec, provider) {
|
|
367
|
+
function speedError (key, speed, spec, provider, adapter) {
|
|
369
368
|
if (!hasSpeed(spec, speed)) {
|
|
370
369
|
const available = speedNames(spec)
|
|
371
370
|
const detail = available.length
|
|
372
371
|
? `Available: ${available.join(', ')}`
|
|
373
372
|
: 'It declares no speed lanes.'
|
|
374
|
-
const
|
|
375
|
-
|
|
376
|
-
|
|
373
|
+
const colon = speed.indexOf(':')
|
|
374
|
+
const hint = colon < 0
|
|
375
|
+
? ''
|
|
376
|
+
: ` Suffix order is ':effort' then '@speed' — did you mean '${key}:${speed.slice(colon + 1)}@${speed.slice(0, colon)}'?`
|
|
377
377
|
return errorEvent(
|
|
378
378
|
`Model '${key}' does not support speed lane '${speed}'. ${detail}${hint}`,
|
|
379
379
|
'SESSION_INVALID_SPEED'
|
|
380
380
|
)
|
|
381
381
|
}
|
|
382
|
-
|
|
382
|
+
const lanes = adapter?.speedLanes
|
|
383
|
+
if (!lanes?.has(speed)) {
|
|
384
|
+
const detail = lanes
|
|
385
|
+
? `It accepts: ${[...lanes].join(', ')}.`
|
|
386
|
+
: 'It implements no speed lanes.'
|
|
383
387
|
return errorEvent(
|
|
384
|
-
`Provider '${provider}'
|
|
388
|
+
`Provider '${provider}' cannot serve speed lane '${speed}', but '${key}' declares it. ${detail}`,
|
|
385
389
|
'SESSION_SPEED_NOT_IMPLEMENTED'
|
|
386
390
|
)
|
|
387
391
|
}
|
|
@@ -456,6 +460,8 @@ function finalizeSpanOk (span, result, sawDelta = false, maxInterFrameMs = 0) {
|
|
|
456
460
|
}
|
|
457
461
|
if (result?.cacheWriteInputTokens) attrs['mohdel.cache_write_input_tokens'] = result.cacheWriteInputTokens
|
|
458
462
|
if (result?.cacheReadInputTokens) attrs['mohdel.cache_read_input_tokens'] = result.cacheReadInputTokens
|
|
463
|
+
if (result?.speed) attrs['mohdel.speed'] = result.speed
|
|
464
|
+
if (result?.servedSpeed !== undefined) attrs['mohdel.served_speed'] = result.servedSpeed ?? 'standard'
|
|
459
465
|
if (result?.cost != null) attrs['mohdel.cost'] = result.cost
|
|
460
466
|
if (result?.warning) attrs['mohdel.warning'] = result.warning
|
|
461
467
|
if (result?.timestamps?.start && result?.timestamps?.first) {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "mohdel",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.121.0",
|
|
4
4
|
"license": "MIT",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "Christophe Le Bars",
|
|
@@ -108,12 +108,12 @@
|
|
|
108
108
|
"@opentelemetry/exporter-trace-otlp-grpc": "^0.221.0",
|
|
109
109
|
"@opentelemetry/sdk-node": "^0.221.0",
|
|
110
110
|
"chalk": "^6.0.0",
|
|
111
|
-
"mohdel-thin-gate-linux-x64-gnu": "0.
|
|
111
|
+
"mohdel-thin-gate-linux-x64-gnu": "0.121.0"
|
|
112
112
|
},
|
|
113
113
|
"dependencies": {
|
|
114
|
-
"@anthropic-ai/sdk": "^0.
|
|
114
|
+
"@anthropic-ai/sdk": "^0.117.1",
|
|
115
115
|
"@cerebras/cerebras_cloud_sdk": "^1.91.0",
|
|
116
|
-
"@google/genai": "^2.
|
|
116
|
+
"@google/genai": "^2.17.1",
|
|
117
117
|
"@opentelemetry/api": "^1.9.1",
|
|
118
118
|
"env-paths": "^4.0.0",
|
|
119
119
|
"groq-sdk": "^1.5.0",
|
|
@@ -126,7 +126,7 @@
|
|
|
126
126
|
"devDependencies": {
|
|
127
127
|
"gpt-tokenizer": "^3.4.0",
|
|
128
128
|
"lint-staged": "^17.3.0",
|
|
129
|
-
"release-it": "^21.0.
|
|
129
|
+
"release-it": "^21.0.2",
|
|
130
130
|
"standard": "^17.1.2",
|
|
131
131
|
"vitest": "^4.1.10"
|
|
132
132
|
}
|
package/src/cli/ask.js
CHANGED
|
@@ -176,6 +176,12 @@ Examples:
|
|
|
176
176
|
if (tokens.outputTokens) summary.push(`${tokens.outputTokens} out`)
|
|
177
177
|
if (tokens.thinkingTokens) summary.push(`${tokens.thinkingTokens} think`)
|
|
178
178
|
if (tokens.cost != null) summary.push(`$${tokens.cost.toFixed(4)}`)
|
|
179
|
+
if (tokens.speed) {
|
|
180
|
+
const served = tokens.servedSpeed
|
|
181
|
+
summary.push(served === tokens.speed
|
|
182
|
+
? `${tokens.speed} lane`
|
|
183
|
+
: `${tokens.speed} lane → ${served ?? 'standard'}`)
|
|
184
|
+
}
|
|
179
185
|
const ts = tokens.timestamps
|
|
180
186
|
if (ts) {
|
|
181
187
|
const toMs = (a, b) => {
|
package/src/cli/check.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { label, err, warn, ok } from './colors.js'
|
|
2
2
|
import providers from '../lib/providers.js'
|
|
3
3
|
import { validate, isValidTag } from '../lib/schema.js'
|
|
4
|
-
import {
|
|
4
|
+
import { adapters } from '../../js/session/adapters/index.js'
|
|
5
5
|
import { getCuratedModels, loadDefaultEnv, catalogEntries, catalogValues } from '../lib/common.js'
|
|
6
6
|
|
|
7
7
|
// --- Local validation ---
|
|
@@ -51,10 +51,12 @@ const checkLocal = (curated) => {
|
|
|
51
51
|
warnings.push(`${key}: has thinkingEffortLevels but no defaultThinkingEffort`)
|
|
52
52
|
}
|
|
53
53
|
|
|
54
|
-
|
|
55
|
-
errors.push(`${key}: declares speeds but provider '${keyProvider}' has no adapter support — calls on those lanes would fail at dispatch`)
|
|
56
|
-
}
|
|
54
|
+
const lanes = adapters[keyProvider]?.speedLanes
|
|
57
55
|
for (const [lane, overlay] of Object.entries(spec.speeds || {})) {
|
|
56
|
+
if (!lanes?.has(lane)) {
|
|
57
|
+
const detail = lanes ? `accepts: ${[...lanes].join(', ')}` : 'implements no speed lanes'
|
|
58
|
+
errors.push(`${key}: speeds.${lane} — provider '${keyProvider}' ${detail}; calls on that lane would fail at dispatch`)
|
|
59
|
+
}
|
|
58
60
|
for (const priceField of ['inputPrice', 'outputPrice', 'thinkingPrice']) {
|
|
59
61
|
const val = overlay[priceField]
|
|
60
62
|
if (val != null && typeof val === 'object' && val.default == null) {
|
package/src/lib/index.js
CHANGED
|
@@ -75,6 +75,17 @@ const resolvePrice = (price, inputTokens) => {
|
|
|
75
75
|
// @internal — exported for unit tests only.
|
|
76
76
|
export { resolvePrice as _resolvePriceForTests }
|
|
77
77
|
|
|
78
|
+
// A lane candidate containing `:` means the suffixes were written the
|
|
79
|
+
// other way round, which is otherwise reported as an unknown lane
|
|
80
|
+
// spelled `fast:none`.
|
|
81
|
+
const reversedSuffixHint = (base, candidate) => {
|
|
82
|
+
const colon = candidate.indexOf(':')
|
|
83
|
+
if (colon < 0) return ''
|
|
84
|
+
const speed = candidate.slice(0, colon)
|
|
85
|
+
const effort = candidate.slice(colon + 1)
|
|
86
|
+
return ` Suffix order is ':effort' then '@speed' — did you mean '${base}:${effort}@${speed}'?`
|
|
87
|
+
}
|
|
88
|
+
|
|
78
89
|
const normalizeModelSpec = (resolvedModelId, modelSpec, providerConfig) => {
|
|
79
90
|
const normalized = { ...modelSpec }
|
|
80
91
|
if (!normalized.provider) {
|
|
@@ -357,9 +368,12 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
|
|
|
357
368
|
if (aliasSpeed && !Object.hasOwn(modelSpec.speeds || {}, aliasSpeed)) {
|
|
358
369
|
const available = Object.keys(modelSpec.speeds || {})
|
|
359
370
|
const detail = available.length
|
|
360
|
-
? `Available: ${available.join(', ')}
|
|
371
|
+
? `Available: ${available.join(', ')}.`
|
|
361
372
|
: 'It declares no speed lanes.'
|
|
362
|
-
throw new Error(
|
|
373
|
+
throw new Error(
|
|
374
|
+
`Model '${resolvedModelId}' does not support speed lane '${aliasSpeed}'. ${detail}` +
|
|
375
|
+
reversedSuffixHint(resolvedModelId, aliasSpeed)
|
|
376
|
+
)
|
|
363
377
|
}
|
|
364
378
|
|
|
365
379
|
// Validate outputEffort alias against model capabilities
|
package/src/lib/schema.js
CHANGED
|
@@ -1,14 +1,11 @@
|
|
|
1
1
|
import { SPEED_OVERRIDABLE } from '../../js/session/adapters/_speed.js'
|
|
2
2
|
|
|
3
3
|
const validateSpeeds = (speeds) => {
|
|
4
|
-
const allowed = new Set(
|
|
4
|
+
const allowed = new Set(SPEED_OVERRIDABLE)
|
|
5
5
|
for (const [lane, overlay] of Object.entries(speeds)) {
|
|
6
6
|
if (typeof overlay !== 'object' || overlay === null || Array.isArray(overlay)) {
|
|
7
7
|
return `lane '${lane}' must be an object`
|
|
8
8
|
}
|
|
9
|
-
if (typeof overlay.wire !== 'string' || !overlay.wire) {
|
|
10
|
-
return `lane '${lane}' must set a string 'wire' value`
|
|
11
|
-
}
|
|
12
9
|
for (const field of Object.keys(overlay)) {
|
|
13
10
|
if (!allowed.has(field)) {
|
|
14
11
|
return `lane '${lane}' may not override '${field}' (allowed: ${[...allowed].join(', ')})`
|
|
@@ -30,6 +27,9 @@ const fieldDefs = {
|
|
|
30
27
|
inputPrice: { type: 'number', altType: 'object' },
|
|
31
28
|
outputPrice: { type: 'number', altType: 'object' },
|
|
32
29
|
thinkingPrice: { type: 'number', altType: 'object' },
|
|
30
|
+
cacheReadPrice: { type: 'number', altType: 'object' },
|
|
31
|
+
cacheWritePrice: { type: 'number', altType: 'object' },
|
|
32
|
+
cacheWrite1hPrice: { type: 'number', altType: 'object' },
|
|
33
33
|
contextTokenLimit: { type: 'number' },
|
|
34
34
|
outputTokenLimit: { type: 'number' },
|
|
35
35
|
thinkingTokenLimit: { type: 'number' },
|