mohdel 0.120.0 → 0.121.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -102,7 +102,7 @@ mo ask anthropic/claude-sonnet-4-6 --stream "write a haiku about recursion"
102
102
  mo ask anthropic/claude-opus-4-6 --effort high "prove P != NP"
103
103
 
104
104
  # On a faster service lane, when the model sells one
105
- mo ask anthropic/claude-opus-4-6@fast "triage this alert"
105
+ mo ask openai/gpt-5.6-luna@fast "triage this alert"
106
106
 
107
107
  # Speech → text from an audio file
108
108
  mo transcribe groq/whisper-large-v3-turbo meeting.mp3
@@ -318,7 +318,7 @@ What each provider supports through mohdel's unified interface:
318
318
 
319
319
  Adapter capability ≠ model capability — whether a given model accepts images, tools, or thinking effort depends on the model spec in `curated.json`. The adapter passes through what the envelope supplies; the provider rejects unsupported combos.
320
320
 
321
- **Service speeds** are the exception to that pass-through rule. Where a provider sells the same weights at several speeds (Anthropic's fast mode, and equivalents elsewhere), the lanes a model sells are declared in its `curated.json` entry under `speeds`, with their own prices and rate limits. Select one with `speed` or the `@lane` id suffix. There is no default lane and no fallback: an undeclared lane fails the call before it is sent, because a model that silently ignores an unsupported lane would otherwise be billed at the lane's rates for standard service. See [docs/CATALOG.md](docs/CATALOG.md#service-speeds).
321
+ **Service speeds** are the exception to that pass-through rule. Where a provider sells the same weights at several speeds (OpenAI's `service_tier`, and equivalents elsewhere), the lanes a model sells are declared in its `curated.json` entry under `speeds`, with their own prices and rate limits. Select one with `speed` or the `@lane` id suffix. There is no default lane and no fallback: an undeclared lane fails the call before it is sent, because a model that silently ignores an unsupported lane would otherwise be billed at the lane's rates for standard service. See [docs/CATALOG.md](docs/CATALOG.md#service-speeds).
322
322
 
323
323
  ## Local Development
324
324
 
@@ -2,7 +2,7 @@
2
2
  "$schema": "https://json-schema.org/draft/2020-12/schema",
3
3
  "$id": "https://github.com/clbrge/mohdel/blob/main/config/curated.schema.json",
4
4
  "title": "Mohdel curated.json",
5
- "description": "Schema for ~/.config/mohdel/curated.json — see docs/CATALOG.md for the full reference.",
5
+ "description": "Schema for ~/.config/mohdel/curated.json \u2014 see docs/CATALOG.md for the full reference.",
6
6
  "type": "object",
7
7
  "propertyNames": {
8
8
  "pattern": "^([a-z0-9_-]+/.+|_[a-zA-Z0-9_-]+)$",
@@ -10,16 +10,25 @@
10
10
  },
11
11
  "additionalProperties": {
12
12
  "oneOf": [
13
- { "$ref": "#/$defs/deprecatedStub" },
14
- { "$ref": "#/$defs/modelEntry" },
15
- { "type": "string", "description": "Inline comment value for underscore-prefixed keys." }
13
+ {
14
+ "$ref": "#/$defs/deprecatedStub"
15
+ },
16
+ {
17
+ "$ref": "#/$defs/modelEntry"
18
+ },
19
+ {
20
+ "type": "string",
21
+ "description": "Inline comment value for underscore-prefixed keys."
22
+ }
16
23
  ]
17
24
  },
18
25
  "$defs": {
19
26
  "deprecatedStub": {
20
27
  "type": "object",
21
28
  "description": "Redirect from a retired model id to its replacement.",
22
- "required": ["deprecated"],
29
+ "required": [
30
+ "deprecated"
31
+ ],
23
32
  "additionalProperties": true,
24
33
  "properties": {
25
34
  "deprecated": {
@@ -31,7 +40,11 @@
31
40
  "modelEntry": {
32
41
  "type": "object",
33
42
  "description": "A real model entry. 'model', 'creator', and 'inputFormat' are required.",
34
- "required": ["model", "creator", "inputFormat"],
43
+ "required": [
44
+ "model",
45
+ "creator",
46
+ "inputFormat"
47
+ ],
35
48
  "additionalProperties": true,
36
49
  "properties": {
37
50
  "model": {
@@ -45,7 +58,15 @@
45
58
  "inputFormat": {
46
59
  "type": "array",
47
60
  "description": "Accepted input modalities. Defaults to ['text'].",
48
- "items": { "type": "string", "enum": ["text", "image", "video", "audio"] },
61
+ "items": {
62
+ "type": "string",
63
+ "enum": [
64
+ "text",
65
+ "image",
66
+ "video",
67
+ "audio"
68
+ ]
69
+ },
49
70
  "minItems": 1,
50
71
  "uniqueItems": true
51
72
  },
@@ -53,142 +74,454 @@
53
74
  "type": "string",
54
75
  "description": "Routing provider. Defaults to the provider segment of the catalog key."
55
76
  },
56
- "sdk": { "type": "string", "description": "SDK adapter to use (some providers require this)." },
77
+ "sdk": {
78
+ "type": "string",
79
+ "description": "SDK adapter to use (some providers require this)."
80
+ },
57
81
  "type": {
58
82
  "type": "string",
59
- "enum": ["model", "image", "transcription"],
83
+ "enum": [
84
+ "model",
85
+ "image",
86
+ "transcription"
87
+ ],
60
88
  "default": "model",
61
89
  "description": "'model' for chat/completion, 'image' for image generation, 'transcription' for speech-to-text."
62
90
  },
63
- "label": { "type": "string", "description": "Human-readable name shown in UIs." },
64
- "description": { "type": "string" },
65
- "version": { "type": "string" },
66
- "createdAt": { "type": "string", "format": "date-time" },
67
- "created": { "type": "number", "description": "Unix timestamp (seconds)." },
68
-
91
+ "label": {
92
+ "type": "string",
93
+ "description": "Human-readable name shown in UIs."
94
+ },
95
+ "description": {
96
+ "type": "string"
97
+ },
98
+ "version": {
99
+ "type": "string"
100
+ },
101
+ "createdAt": {
102
+ "type": "string",
103
+ "format": "date-time"
104
+ },
105
+ "created": {
106
+ "type": "number",
107
+ "description": "Unix timestamp (seconds)."
108
+ },
69
109
  "inputPrice": {
70
110
  "oneOf": [
71
- { "type": "number", "minimum": 0 },
72
- { "type": "object", "required": ["default"], "additionalProperties": { "type": "number" }, "properties": { "default": { "type": "number" } } }
111
+ {
112
+ "type": "number",
113
+ "minimum": 0
114
+ },
115
+ {
116
+ "type": "object",
117
+ "required": [
118
+ "default"
119
+ ],
120
+ "additionalProperties": {
121
+ "type": "number"
122
+ },
123
+ "properties": {
124
+ "default": {
125
+ "type": "number"
126
+ }
127
+ }
128
+ }
73
129
  ],
74
- "description": "USD per 1M input tokens. Object form is for tiered pricing — must include a 'default' key."
130
+ "description": "USD per 1M input tokens. Object form is for tiered pricing \u2014 must include a 'default' key."
75
131
  },
76
132
  "outputPrice": {
77
133
  "oneOf": [
78
- { "type": "number", "minimum": 0 },
79
- { "type": "object", "required": ["default"], "additionalProperties": { "type": "number" }, "properties": { "default": { "type": "number" } } }
134
+ {
135
+ "type": "number",
136
+ "minimum": 0
137
+ },
138
+ {
139
+ "type": "object",
140
+ "required": [
141
+ "default"
142
+ ],
143
+ "additionalProperties": {
144
+ "type": "number"
145
+ },
146
+ "properties": {
147
+ "default": {
148
+ "type": "number"
149
+ }
150
+ }
151
+ }
80
152
  ],
81
153
  "description": "USD per 1M output tokens."
82
154
  },
83
155
  "thinkingPrice": {
84
156
  "oneOf": [
85
- { "type": "number", "minimum": 0 },
86
- { "type": "object", "required": ["default"], "additionalProperties": { "type": "number" }, "properties": { "default": { "type": "number" } } }
157
+ {
158
+ "type": "number",
159
+ "minimum": 0
160
+ },
161
+ {
162
+ "type": "object",
163
+ "required": [
164
+ "default"
165
+ ],
166
+ "additionalProperties": {
167
+ "type": "number"
168
+ },
169
+ "properties": {
170
+ "default": {
171
+ "type": "number"
172
+ }
173
+ }
174
+ }
87
175
  ],
88
176
  "description": "USD per 1M thinking/reasoning tokens (when the provider bills these separately)."
89
177
  },
90
- "cacheWritePrice": { "type": "number", "minimum": 0, "description": "USD per 1M tokens written to provider-side prompt cache." },
91
- "cacheWrite1hPrice": { "type": "number", "minimum": 0, "description": "USD per 1M tokens written with a 1h TTL, when the provider prices that above the default write rate. Falls back to cacheWritePrice." },
92
- "cacheReadPrice": { "type": "number", "minimum": 0, "description": "USD per 1M tokens served from provider-side prompt cache." },
93
-
94
- "contextTokenLimit": { "type": "integer", "minimum": 1, "description": "Maximum total tokens (input + output)." },
95
- "outputTokenLimit": { "type": "integer", "minimum": 1, "description": "Maximum output tokens per call." },
96
- "thinkingTokenLimit": { "type": "integer", "minimum": 1, "description": "Maximum thinking tokens per call (when separate from output)." },
97
- "tokenizerHeadroom": { "type": "number", "exclusiveMinimum": 0, "description": "Multiplier applied to local token estimates to account for tokenizer drift." },
98
-
178
+ "cacheWritePrice": {
179
+ "oneOf": [
180
+ {
181
+ "type": "number",
182
+ "minimum": 0
183
+ },
184
+ {
185
+ "type": "object",
186
+ "required": [
187
+ "default"
188
+ ],
189
+ "additionalProperties": {
190
+ "type": "number"
191
+ },
192
+ "properties": {
193
+ "default": {
194
+ "type": "number"
195
+ }
196
+ }
197
+ }
198
+ ],
199
+ "description": "USD per 1M tokens written to provider-side prompt cache. Object form is tiered on input tokens."
200
+ },
201
+ "cacheWrite1hPrice": {
202
+ "oneOf": [
203
+ {
204
+ "type": "number",
205
+ "minimum": 0
206
+ },
207
+ {
208
+ "type": "object",
209
+ "required": [
210
+ "default"
211
+ ],
212
+ "additionalProperties": {
213
+ "type": "number"
214
+ },
215
+ "properties": {
216
+ "default": {
217
+ "type": "number"
218
+ }
219
+ }
220
+ }
221
+ ],
222
+ "description": "USD per 1M tokens written with a 1h TTL, when the provider prices that above the default write rate. Falls back to cacheWritePrice."
223
+ },
224
+ "cacheReadPrice": {
225
+ "oneOf": [
226
+ {
227
+ "type": "number",
228
+ "minimum": 0
229
+ },
230
+ {
231
+ "type": "object",
232
+ "required": [
233
+ "default"
234
+ ],
235
+ "additionalProperties": {
236
+ "type": "number"
237
+ },
238
+ "properties": {
239
+ "default": {
240
+ "type": "number"
241
+ }
242
+ }
243
+ }
244
+ ],
245
+ "description": "USD per 1M tokens served from provider-side prompt cache. Object form is tiered on input tokens."
246
+ },
247
+ "contextTokenLimit": {
248
+ "type": "integer",
249
+ "minimum": 1,
250
+ "description": "Maximum total tokens (input + output)."
251
+ },
252
+ "outputTokenLimit": {
253
+ "type": "integer",
254
+ "minimum": 1,
255
+ "description": "Maximum output tokens per call."
256
+ },
257
+ "thinkingTokenLimit": {
258
+ "type": "integer",
259
+ "minimum": 1,
260
+ "description": "Maximum thinking tokens per call (when separate from output)."
261
+ },
262
+ "tokenizerHeadroom": {
263
+ "type": "number",
264
+ "exclusiveMinimum": 0,
265
+ "description": "Multiplier applied to local token estimates to account for tokenizer drift."
266
+ },
99
267
  "thinkingEffortLevels": {
100
268
  "oneOf": [
101
- { "type": "null" },
269
+ {
270
+ "type": "null"
271
+ },
102
272
  {
103
273
  "type": "object",
104
- "description": "Map effort name → provider-native budget. Standard names: low/medium/high/xhigh/max/none.",
105
- "additionalProperties": { "type": "number", "minimum": 0 }
274
+ "description": "Map effort name \u2192 provider-native budget. Standard names: low/medium/high/xhigh/max/none.",
275
+ "additionalProperties": {
276
+ "type": "number",
277
+ "minimum": 0
278
+ }
106
279
  }
107
280
  ]
108
281
  },
109
- "defaultThinkingEffort": { "type": "string", "description": "Effort level used when the envelope omits 'outputEffort'." },
110
-
282
+ "defaultThinkingEffort": {
283
+ "type": "string",
284
+ "description": "Effort level used when the envelope omits 'outputEffort'."
285
+ },
111
286
  "speeds": {
112
287
  "type": "object",
113
- "description": "Service speed lanes this model sells, keyed by lane name. A lane is a provider request parameter buying a different speed/price point for the same weights. There is no default lane: omitting 'speed' on the envelope sends no parameter. Only lanes declared here may be requested.",
288
+ "description": "Service speed lanes this model sells, keyed by lane name. A lane is a provider request parameter buying a different speed/price point for the same weights. There is no default lane: omitting 'speed' on the envelope sends no parameter. Only lanes declared here may be requested, and only names the provider's adapter accepts.",
114
289
  "additionalProperties": {
115
290
  "type": "object",
116
- "required": ["wire"],
117
291
  "additionalProperties": false,
118
- "description": "Lane overlay. Fields named here override the base entry; unnamed fields fall through. Only prices and rate limits are overridable — anything that would change what the model is belongs in its own entry.",
292
+ "description": "Lane overlay: the price and rate-limit fields that differ from the base entry. Unnamed fields fall through. Lane names come from the provider's own vocabulary (OpenAI: fast, priority, flex, scale) \u2014 the adapter decides how a name reaches the wire.",
119
293
  "properties": {
120
- "wire": { "type": "string", "description": "Provider-native value for the lane parameter (e.g. 'fast')." },
121
294
  "inputPrice": {
122
295
  "oneOf": [
123
- { "type": "number", "minimum": 0 },
124
- { "type": "object", "required": ["default"], "additionalProperties": { "type": "number" }, "properties": { "default": { "type": "number" } } }
296
+ {
297
+ "type": "number",
298
+ "minimum": 0
299
+ },
300
+ {
301
+ "type": "object",
302
+ "required": [
303
+ "default"
304
+ ],
305
+ "additionalProperties": {
306
+ "type": "number"
307
+ },
308
+ "properties": {
309
+ "default": {
310
+ "type": "number"
311
+ }
312
+ }
313
+ }
125
314
  ]
126
315
  },
127
316
  "outputPrice": {
128
317
  "oneOf": [
129
- { "type": "number", "minimum": 0 },
130
- { "type": "object", "required": ["default"], "additionalProperties": { "type": "number" }, "properties": { "default": { "type": "number" } } }
318
+ {
319
+ "type": "number",
320
+ "minimum": 0
321
+ },
322
+ {
323
+ "type": "object",
324
+ "required": [
325
+ "default"
326
+ ],
327
+ "additionalProperties": {
328
+ "type": "number"
329
+ },
330
+ "properties": {
331
+ "default": {
332
+ "type": "number"
333
+ }
334
+ }
335
+ }
131
336
  ]
132
337
  },
133
338
  "thinkingPrice": {
134
339
  "oneOf": [
135
- { "type": "number", "minimum": 0 },
136
- { "type": "object", "required": ["default"], "additionalProperties": { "type": "number" }, "properties": { "default": { "type": "number" } } }
340
+ {
341
+ "type": "number",
342
+ "minimum": 0
343
+ },
344
+ {
345
+ "type": "object",
346
+ "required": [
347
+ "default"
348
+ ],
349
+ "additionalProperties": {
350
+ "type": "number"
351
+ },
352
+ "properties": {
353
+ "default": {
354
+ "type": "number"
355
+ }
356
+ }
357
+ }
358
+ ]
359
+ },
360
+ "cacheWritePrice": {
361
+ "oneOf": [
362
+ {
363
+ "type": "number",
364
+ "minimum": 0
365
+ },
366
+ {
367
+ "type": "object",
368
+ "required": [
369
+ "default"
370
+ ],
371
+ "additionalProperties": {
372
+ "type": "number"
373
+ },
374
+ "properties": {
375
+ "default": {
376
+ "type": "number"
377
+ }
378
+ }
379
+ }
137
380
  ]
138
381
  },
139
- "cacheWritePrice": { "type": "number", "minimum": 0 },
140
- "cacheWrite1hPrice": { "type": "number", "minimum": 0 },
141
- "cacheReadPrice": { "type": "number", "minimum": 0 },
142
- "rpmLimit": { "type": "integer", "minimum": 1 },
143
- "tpmLimit": { "type": "integer", "minimum": 1 }
382
+ "cacheWrite1hPrice": {
383
+ "oneOf": [
384
+ {
385
+ "type": "number",
386
+ "minimum": 0
387
+ },
388
+ {
389
+ "type": "object",
390
+ "required": [
391
+ "default"
392
+ ],
393
+ "additionalProperties": {
394
+ "type": "number"
395
+ },
396
+ "properties": {
397
+ "default": {
398
+ "type": "number"
399
+ }
400
+ }
401
+ }
402
+ ]
403
+ },
404
+ "cacheReadPrice": {
405
+ "oneOf": [
406
+ {
407
+ "type": "number",
408
+ "minimum": 0
409
+ },
410
+ {
411
+ "type": "object",
412
+ "required": [
413
+ "default"
414
+ ],
415
+ "additionalProperties": {
416
+ "type": "number"
417
+ },
418
+ "properties": {
419
+ "default": {
420
+ "type": "number"
421
+ }
422
+ }
423
+ }
424
+ ]
425
+ },
426
+ "rpmLimit": {
427
+ "type": "integer",
428
+ "minimum": 1
429
+ },
430
+ "tpmLimit": {
431
+ "type": "integer",
432
+ "minimum": 1
433
+ }
144
434
  }
145
435
  }
146
436
  },
147
-
148
437
  "tags": {
149
438
  "type": "array",
150
- "items": { "type": "string", "pattern": "^[a-zA-Z][a-zA-Z0-9._-]{0,31}$" },
439
+ "items": {
440
+ "type": "string",
441
+ "pattern": "^[a-zA-Z][a-zA-Z0-9._-]{0,31}$"
442
+ },
151
443
  "uniqueItems": true,
152
444
  "description": "Free-form classification tags (used by 'mo bench --tag', 'mo rank', and caller-side selection)."
153
445
  },
154
446
  "aliases": {
155
447
  "type": "array",
156
- "items": { "type": "string" },
448
+ "items": {
449
+ "type": "string"
450
+ },
157
451
  "description": "Alternative ids that should resolve to this entry."
158
452
  },
159
453
  "replaces": {
160
454
  "type": "array",
161
- "items": { "type": "string" },
455
+ "items": {
456
+ "type": "string"
457
+ },
162
458
  "description": "Older model ids this one supersedes."
163
459
  },
164
460
  "leaderboard": {
165
461
  "type": "array",
166
- "items": { "type": "number" },
462
+ "items": {
463
+ "type": "number"
464
+ },
167
465
  "minItems": 3,
168
466
  "maxItems": 3,
169
- "description": "[intelligence, speed, latency] triple — drives 'mo rank'."
170
- },
171
- "leaderboardNote": { "type": "string" },
172
-
173
- "supportsTools": { "type": "boolean", "description": "Set false to mark a model as tool-less." },
174
-
175
- "imagePrice": { "type": "number", "minimum": 0, "description": "USD per generated image (image-type entries)." },
176
- "imageEndpoint": { "type": "string", "description": "Provider-side image endpoint name." },
177
- "imageDefaultSize": { "type": "string", "description": "Default size when envelope omits one (e.g. '1024x1024')." },
178
- "transcriptionPrice": { "type": "number", "minimum": 0, "description": "USD per audio minute (transcription-type entries). Token-billed transcription models (OpenAI gpt-4o-*-transcribe) use inputPrice/outputPrice instead." },
179
-
180
- "rpmLimit": { "type": "integer", "minimum": 1, "description": "Requests per minute. Overrides provider default." },
181
- "tpmLimit": { "type": "integer", "minimum": 1, "description": "Tokens per minute. Overrides provider default." },
467
+ "description": "[intelligence, speed, latency] triple \u2014 drives 'mo rank'."
468
+ },
469
+ "leaderboardNote": {
470
+ "type": "string"
471
+ },
472
+ "supportsTools": {
473
+ "type": "boolean",
474
+ "description": "Set false to mark a model as tool-less."
475
+ },
476
+ "imagePrice": {
477
+ "type": "number",
478
+ "minimum": 0,
479
+ "description": "USD per generated image (image-type entries)."
480
+ },
481
+ "imageEndpoint": {
482
+ "type": "string",
483
+ "description": "Provider-side image endpoint name."
484
+ },
485
+ "imageDefaultSize": {
486
+ "type": "string",
487
+ "description": "Default size when envelope omits one (e.g. '1024x1024')."
488
+ },
489
+ "transcriptionPrice": {
490
+ "type": "number",
491
+ "minimum": 0,
492
+ "description": "USD per audio minute (transcription-type entries). Token-billed transcription models (OpenAI gpt-4o-*-transcribe) use inputPrice/outputPrice instead."
493
+ },
494
+ "rpmLimit": {
495
+ "type": "integer",
496
+ "minimum": 1,
497
+ "description": "Requests per minute. Overrides provider default."
498
+ },
499
+ "tpmLimit": {
500
+ "type": "integer",
501
+ "minimum": 1,
502
+ "description": "Tokens per minute. Overrides provider default."
503
+ },
182
504
  "rateLimitScope": {
183
505
  "type": "string",
184
- "enum": ["model", "provider"],
506
+ "enum": [
507
+ "model",
508
+ "provider"
509
+ ],
185
510
  "description": "'model' = private budget. 'provider' = shared with provider-level pool."
186
511
  },
187
-
188
- "deprecated": { "type": "string", "description": "If present, this entry is treated as a stub (use 'deprecatedStub' shape)." },
189
- "suspended": { "type": "string", "description": "Reason this model is temporarily disabled." },
190
-
191
- "displayName": { "type": "string", "deprecated": true, "description": "Deprecated — use 'label'." }
512
+ "deprecated": {
513
+ "type": "string",
514
+ "description": "If present, this entry is treated as a stub (use 'deprecatedStub' shape)."
515
+ },
516
+ "suspended": {
517
+ "type": "string",
518
+ "description": "Reason this model is temporarily disabled."
519
+ },
520
+ "displayName": {
521
+ "type": "string",
522
+ "deprecated": true,
523
+ "description": "Deprecated \u2014 use 'label'."
524
+ }
192
525
  }
193
526
  }
194
527
  }
@@ -19,7 +19,6 @@
19
19
  import { getSpec } from './_catalog.js'
20
20
  import { classifyProviderError } from './_errors.js'
21
21
  import { costFor } from './_pricing.js'
22
- import { applySpeed } from './_speed.js'
23
22
  import { catalogKey, bareOf } from '#core/model-id.js'
24
23
  import {
25
24
  STATUS_COMPLETED,
@@ -371,8 +370,6 @@ function buildRequest (envelope, spec, config) {
371
370
  args[config.identifierField || 'user'] = envelope.identifier
372
371
  }
373
372
 
374
- applySpeed(args, envelope, spec)
375
-
376
373
  return args
377
374
  }
378
375
 
@@ -1,30 +1,21 @@
1
1
  /**
2
- * Service-speed lanes.
2
+ * Service-speed lanes — catalog side.
3
3
  *
4
- * A lane is a provider request parameter that buys a different
5
- * speed/price point for the same weights (Anthropic `speed: "fast"`).
6
- * Lanes are declared per catalog entry under `speeds`, keyed by lane
7
- * name, each carrying the wire value plus any price and rate-limit
8
- * fields that differ from the base entry.
4
+ * A lane is a named service speed a model sells, declared per catalog
5
+ * entry under `speeds` with the price and rate-limit fields that
6
+ * differ from the base entry. This module owns only what is common to
7
+ * every provider: what the entry declares, and what that means for
8
+ * pricing and throttling.
9
9
  *
10
- * Two declarations must agree for a lane to be usable: the entry
11
- * declares it (`spec.speeds`) and the provider's adapter can emit it
12
- * (`SPEED_PARAMS`). `run.js` checks both before dispatch.
10
+ * How a lane reaches the wire, and how the served lane is read back,
11
+ * is provider protocol and lives in the adapter. An adapter that
12
+ * handles lanes advertises the names it accepts on `speedLanes`;
13
+ * absence of that property is the declaration that it handles none,
14
+ * and is what `run.js` checks before dispatch.
13
15
  *
14
16
  * @module session/adapters/_speed
15
17
  */
16
18
 
17
- import { providerOf } from '#core/model-id.js'
18
-
19
- /**
20
- * Providers whose adapter emits a lane parameter, mapped to the
21
- * native parameter name. Absence from this table is the declaration
22
- * that a provider has no lanes.
23
- */
24
- export const SPEED_PARAMS = Object.freeze({
25
- anthropic: 'speed'
26
- })
27
-
28
19
  /** Spec fields a lane overlay may restate. */
29
20
  export const SPEED_OVERRIDABLE = Object.freeze([
30
21
  'inputPrice',
@@ -37,25 +28,9 @@ export const SPEED_OVERRIDABLE = Object.freeze([
37
28
  'tpmLimit'
38
29
  ])
39
30
 
40
- /**
41
- * @param {string} provider
42
- * @returns {boolean}
43
- */
44
- export function providerSupportsSpeed (provider) {
45
- return Object.hasOwn(SPEED_PARAMS, provider)
46
- }
47
-
48
- /**
49
- * @param {string} provider
50
- * @returns {string | undefined}
51
- */
52
- export function speedParamFor (provider) {
53
- return SPEED_PARAMS[provider]
54
- }
55
-
56
31
  /**
57
32
  * Whether `spec` declares `speed`. Uses `hasOwn` rather than
58
- * truthiness so an overlay that only carries `wire` still counts.
33
+ * truthiness so a lane sold at base prices still counts.
59
34
  *
60
35
  * @param {any} spec
61
36
  * @param {string} speed
@@ -73,6 +48,22 @@ export function speedNames (spec) {
73
48
  return spec?.speeds ? Object.keys(spec.speeds) : []
74
49
  }
75
50
 
51
+ /**
52
+ * Whether the lane declares a quota of its own, which is what earns it
53
+ * a private rate-limit bucket. Lanes that don't (OpenAI's service_tier
54
+ * shares the model's TPM/RPM pool) count against the base bucket —
55
+ * giving them their own would silently double the allowance.
56
+ *
57
+ * @param {any} spec
58
+ * @param {string} [speed]
59
+ * @returns {boolean}
60
+ */
61
+ export function speedHasOwnQuota (spec, speed) {
62
+ if (!speed || !hasSpeed(spec, speed)) return false
63
+ const overlay = spec.speeds[speed]
64
+ return overlay.rpmLimit != null || overlay.tpmLimit != null
65
+ }
66
+
76
67
  /**
77
68
  * Spec with `speed`'s overlay applied. Unnamed fields fall through to
78
69
  * the base entry.
@@ -94,26 +85,3 @@ export function mergeSpeed (spec, speed) {
94
85
  }
95
86
  return merged
96
87
  }
97
-
98
- /**
99
- * Set the provider-native lane parameter on an outbound request.
100
- * No-op when the envelope carries no lane.
101
- *
102
- * @param {Record<string, any>} request
103
- * @param {import('#core/envelope.js').CallEnvelope} envelope
104
- * @param {any} spec
105
- * @throws when the envelope carries a lane the provider cannot emit
106
- * or the spec does not declare
107
- */
108
- export function applySpeed (request, envelope, spec) {
109
- if (!envelope.speed) return
110
- const provider = providerOf(envelope.model)
111
- const param = speedParamFor(provider)
112
- if (!param) {
113
- throw new Error(`provider '${provider}' does not implement speed lanes`)
114
- }
115
- if (!hasSpeed(spec, envelope.speed)) {
116
- throw new Error(`model does not declare speed lane '${envelope.speed}'`)
117
- }
118
- request[param] = spec.speeds[envelope.speed].wire
119
- }
@@ -29,7 +29,6 @@ import { classifyProviderError } from './_errors.js'
29
29
  import { loadImages } from './_images.js'
30
30
  import { isTrustedMedia } from './_media.js'
31
31
  import { costFor } from './_pricing.js'
32
- import { applySpeed } from './_speed.js'
33
32
  import { catalogKey, bareOf } from '#core/model-id.js'
34
33
  import {
35
34
  toAnthropicTools,
@@ -353,7 +352,6 @@ function buildRequest (envelope, conversation, system, conversationCacheTtl = nu
353
352
  }
354
353
  }
355
354
 
356
- applySpeed(request, envelope, spec)
357
355
  applyCacheBreakpoints(request, conversationCacheTtl)
358
356
 
359
357
  return request
@@ -31,7 +31,6 @@ import { loadImages } from './_images.js'
31
31
  import { isTrustedMedia } from './_media.js'
32
32
  import { loadVideos } from './_videos.js'
33
33
  import { costFor } from './_pricing.js'
34
- import { applySpeed } from './_speed.js'
35
34
  import { catalogKey, bareOf } from '#core/model-id.js'
36
35
  import {
37
36
  toGeminiTools,
@@ -272,7 +271,6 @@ function buildRequest (envelope, contents, systemInstruction) {
272
271
  contents
273
272
  }
274
273
  if (Object.keys(config).length > 0) request.config = config
275
- applySpeed(request, envelope, spec)
276
274
  return request
277
275
  }
278
276
 
@@ -334,7 +332,12 @@ function buildContents (prompt) {
334
332
  /** @param {string | import('#core/envelope.js').MessagePart[]} content */
335
333
  function safeParseToolResult (content) {
336
334
  const text = flattenText(content)
337
- try { return JSON.parse(text) } catch { return { result: text } }
335
+ let value
336
+ try { value = JSON.parse(text) } catch { return { result: text } }
337
+ // `response` must be a protobuf Struct — a JSON object. A result that
338
+ // parses to a number, string, boolean, null or array is rejected 400.
339
+ if (value === null || typeof value !== 'object' || Array.isArray(value)) return { result: value }
340
+ return value
338
341
  }
339
342
 
340
343
  /** @param {string} role */
@@ -30,7 +30,6 @@ import { classifyProviderError } from './_errors.js'
30
30
  import { loadImages } from './_images.js'
31
31
  import { isTrustedMedia } from './_media.js'
32
32
  import { costFor } from './_pricing.js'
33
- import { applySpeed } from './_speed.js'
34
33
  import { catalogKey, providerOf, bareOf } from '#core/model-id.js'
35
34
  import {
36
35
  toOpenAITools,
@@ -84,6 +83,8 @@ export async function * openai (envelope, deps = {}) {
84
83
  let status = STATUS_COMPLETED
85
84
  /** @type {string | undefined} */
86
85
  let warning
86
+ /** @type {string | null | undefined} */
87
+ let servedTier
87
88
 
88
89
  // Tool-call accumulation: itemId → {call_id, name, arguments}
89
90
  /** @type {Map<string, {call_id: string, name: string, arguments: string}>} */
@@ -131,6 +132,7 @@ export async function * openai (envelope, deps = {}) {
131
132
  break
132
133
 
133
134
  case 'response.completed':
135
+ servedTier = event.response?.service_tier ?? servedTier
134
136
  if (event.response?.usage) {
135
137
  inputTokens = event.response.usage.input_tokens ?? 0
136
138
  outputTokens = event.response.usage.output_tokens ?? 0
@@ -144,6 +146,7 @@ export async function * openai (envelope, deps = {}) {
144
146
  break
145
147
 
146
148
  case 'response.incomplete':
149
+ servedTier = event.response?.service_tier ?? servedTier
147
150
  status = STATUS_INCOMPLETE
148
151
  if (event.response?.incomplete_details?.reason === 'max_output_tokens') {
149
152
  warning = WARNING_INSUFFICIENT_OUTPUT_BUDGET
@@ -190,6 +193,8 @@ export async function * openai (envelope, deps = {}) {
190
193
  // simpler with the additive shape.
191
194
  const regularInputTokens = Math.max(0, inputTokens - cachedInputTokens - cacheWriteTokens)
192
195
 
196
+ const billed = billedEnvelope(envelope, servedTier, log)
197
+
193
198
  /** @type {import('#core/events.js').DoneEvent} */
194
199
  const done = {
195
200
  type: 'done',
@@ -201,8 +206,9 @@ export async function * openai (envelope, deps = {}) {
201
206
  thinkingTokens,
202
207
  ...(cacheWriteTokens > 0 && { cacheWriteInputTokens: cacheWriteTokens }),
203
208
  ...(cachedInputTokens > 0 && { cacheReadInputTokens: cachedInputTokens }),
209
+ ...(envelope.speed && { servedSpeed: billed.served }),
204
210
  cost: costFor(
205
- envelope,
211
+ billed.envelope,
206
212
  {
207
213
  inputTokens: regularInputTokens,
208
214
  outputTokens: messageOutputTokens,
@@ -221,6 +227,66 @@ export async function * openai (envelope, deps = {}) {
221
227
  yield done
222
228
  }
223
229
 
230
+ /**
231
+ * Lane names the OpenAI adapter can put on `service_tier`. `run.js`
232
+ * reads this before dispatch; a lane outside it never reaches here.
233
+ */
234
+ openai.speedLanes = new Set(['fast', 'priority', 'flex', 'scale'])
235
+
236
+ /**
237
+ * Lane the response says was served, given the one requested.
238
+ *
239
+ * OpenAI answers `service_tier: 'priority'` to a granted request for
240
+ * either premium lane, so the echo alone cannot name which was asked
241
+ * for — the request is the other half of the answer. Any other value
242
+ * is the tier that actually ran, `null` when nothing was reported.
243
+ *
244
+ * @param {string} requested
245
+ * @param {string | null | undefined} servedTier
246
+ * @returns {string | null | undefined}
247
+ */
248
+ function servedLane (requested, servedTier) {
249
+ if (servedTier == null) return undefined
250
+ if (servedTier === 'priority') {
251
+ return (requested === 'fast' || requested === 'priority') ? requested : 'priority'
252
+ }
253
+ return openai.speedLanes.has(servedTier) ? servedTier : null
254
+ }
255
+
256
+ /**
257
+ * Envelope to price the call against. OpenAI serves Standard instead
258
+ * when a premium lane is unavailable — documented behaviour above the
259
+ * ramp rate limit — so billing the requested lane would charge premium
260
+ * rates for standard service.
261
+ *
262
+ * When a lane was requested and nothing was reported back, the request
263
+ * is the only evidence available: bill it and say so.
264
+ *
265
+ * @param {import('#core/envelope.js').CallEnvelope} envelope
266
+ * @param {string | null | undefined} servedTier
267
+ * @param {any} [log]
268
+ * @returns {{envelope: import('#core/envelope.js').CallEnvelope, served: string | null}}
269
+ */
270
+ function billedEnvelope (envelope, servedTier, log) {
271
+ if (!envelope.speed) return { envelope, served: null }
272
+
273
+ const served = servedLane(envelope.speed, servedTier)
274
+ if (served === undefined) {
275
+ log?.warn(
276
+ { model: envelope.model, speed: envelope.speed },
277
+ '[mohdel:openai] no service_tier reported; billing the requested lane'
278
+ )
279
+ return { envelope, served: envelope.speed }
280
+ }
281
+ if (served === envelope.speed) return { envelope, served }
282
+
283
+ log?.warn(
284
+ { model: envelope.model, requested: envelope.speed, servedTier, served },
285
+ '[mohdel:openai] provider served a different speed lane; billing what was served'
286
+ )
287
+ return { envelope: { ...envelope, speed: served ?? undefined }, served }
288
+ }
289
+
224
290
  /**
225
291
  * @param {import('#core/envelope.js').CallEnvelope} envelope
226
292
  * @param {Array<any>} input
@@ -288,7 +354,7 @@ function buildRequest (envelope, input, instructions) {
288
354
  }
289
355
  }
290
356
 
291
- applySpeed(request, envelope, spec)
357
+ if (envelope.speed) request.service_tier = envelope.speed
292
358
 
293
359
  return request
294
360
  }
package/js/session/run.js CHANGED
@@ -26,7 +26,7 @@ import { getAdapter } from './adapters/index.js'
26
26
  import { isImageProvider } from './adapters/image/index.js'
27
27
  import { getSpec } from './adapters/_catalog.js'
28
28
  import { getProviderLimits } from './adapters/_providers.js'
29
- import { hasSpeed, mergeSpeed, providerSupportsSpeed, speedNames } from './adapters/_speed.js'
29
+ import { hasSpeed, mergeSpeed, speedHasOwnQuota, speedNames } from './adapters/_speed.js'
30
30
  import { providerOf, catalogKey, effortOf, speedOf } from '#core/model-id.js'
31
31
  import * as defaultCooldown from './_cooldown.js'
32
32
  import * as defaultLimiter from './_rate_limiter.js'
@@ -128,7 +128,7 @@ export async function * run (envelope, {
128
128
  }
129
129
 
130
130
  if (envelope.speed) {
131
- const speedErr = speedError(key, envelope.speed, spec, provider)
131
+ const speedErr = speedError(key, envelope.speed, spec, provider, adapter)
132
132
  if (speedErr) {
133
133
  log.warn({ provider, speed: envelope.speed }, '[mohdel:answer] unusable speed lane')
134
134
  endSpanError(span, new Error(speedErr.error.message))
@@ -150,12 +150,10 @@ export async function * run (envelope, {
150
150
  const providerCfg = resolveProviderLimits(provider) || {}
151
151
  const rpmLimit = effective?.rpmLimit ?? providerCfg.rpmLimit
152
152
  const tpmLimit = effective?.tpmLimit ?? providerCfg.tpmLimit
153
- // An active lane is its own capacity pool, so it gets its own bucket
154
- // whatever `rateLimitScope` says — sharing one would let standard
155
- // traffic throttle the lane being paid for.
156
- const bucketKey = envelope.speed
153
+ const baseBucket = (spec?.rateLimitScope === 'model') ? key : provider
154
+ const bucketKey = speedHasOwnQuota(spec, envelope.speed)
157
155
  ? `${key}@${envelope.speed}`
158
- : (spec?.rateLimitScope === 'model' ? key : provider)
156
+ : baseBucket
159
157
 
160
158
  // `0` is a killswitch ("deny all"), not "unset"; `undefined`/`null`
161
159
  // means no limit configured for that dimension. Gate on nullability
@@ -363,25 +361,31 @@ function normalizeModelId (envelope, resolveSpec) {
363
361
  * @param {string} speed
364
362
  * @param {any} spec
365
363
  * @param {string} provider
364
+ * @param {any} adapter
366
365
  * @returns {import('#core/events.js').ErrorEvent | undefined}
367
366
  */
368
- function speedError (key, speed, spec, provider) {
367
+ function speedError (key, speed, spec, provider, adapter) {
369
368
  if (!hasSpeed(spec, speed)) {
370
369
  const available = speedNames(spec)
371
370
  const detail = available.length
372
371
  ? `Available: ${available.join(', ')}`
373
372
  : 'It declares no speed lanes.'
374
- const hint = speed.includes(':')
375
- ? " Suffix order is ':effort' then '@speed'."
376
- : ''
373
+ const colon = speed.indexOf(':')
374
+ const hint = colon < 0
375
+ ? ''
376
+ : ` Suffix order is ':effort' then '@speed' — did you mean '${key}:${speed.slice(colon + 1)}@${speed.slice(0, colon)}'?`
377
377
  return errorEvent(
378
378
  `Model '${key}' does not support speed lane '${speed}'. ${detail}${hint}`,
379
379
  'SESSION_INVALID_SPEED'
380
380
  )
381
381
  }
382
- if (!providerSupportsSpeed(provider)) {
382
+ const lanes = adapter?.speedLanes
383
+ if (!lanes?.has(speed)) {
384
+ const detail = lanes
385
+ ? `It accepts: ${[...lanes].join(', ')}.`
386
+ : 'It implements no speed lanes.'
383
387
  return errorEvent(
384
- `Provider '${provider}' does not implement speed lanes, but '${key}' declares '${speed}'.`,
388
+ `Provider '${provider}' cannot serve speed lane '${speed}', but '${key}' declares it. ${detail}`,
385
389
  'SESSION_SPEED_NOT_IMPLEMENTED'
386
390
  )
387
391
  }
@@ -456,6 +460,8 @@ function finalizeSpanOk (span, result, sawDelta = false, maxInterFrameMs = 0) {
456
460
  }
457
461
  if (result?.cacheWriteInputTokens) attrs['mohdel.cache_write_input_tokens'] = result.cacheWriteInputTokens
458
462
  if (result?.cacheReadInputTokens) attrs['mohdel.cache_read_input_tokens'] = result.cacheReadInputTokens
463
+ if (result?.speed) attrs['mohdel.speed'] = result.speed
464
+ if (result?.servedSpeed !== undefined) attrs['mohdel.served_speed'] = result.servedSpeed ?? 'standard'
459
465
  if (result?.cost != null) attrs['mohdel.cost'] = result.cost
460
466
  if (result?.warning) attrs['mohdel.warning'] = result.warning
461
467
  if (result?.timestamps?.start && result?.timestamps?.first) {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mohdel",
3
- "version": "0.120.0",
3
+ "version": "0.121.1",
4
4
  "license": "MIT",
5
5
  "author": {
6
6
  "name": "Christophe Le Bars",
@@ -108,16 +108,16 @@
108
108
  "@opentelemetry/exporter-trace-otlp-grpc": "^0.221.0",
109
109
  "@opentelemetry/sdk-node": "^0.221.0",
110
110
  "chalk": "^6.0.0",
111
- "mohdel-thin-gate-linux-x64-gnu": "0.120.0"
111
+ "mohdel-thin-gate-linux-x64-gnu": "0.121.1"
112
112
  },
113
113
  "dependencies": {
114
- "@anthropic-ai/sdk": "^0.115.0",
114
+ "@anthropic-ai/sdk": "^0.120.0",
115
115
  "@cerebras/cerebras_cloud_sdk": "^1.91.0",
116
- "@google/genai": "^2.16.0",
116
+ "@google/genai": "^2.18.0",
117
117
  "@opentelemetry/api": "^1.9.1",
118
118
  "env-paths": "^4.0.0",
119
119
  "groq-sdk": "^1.5.0",
120
- "openai": "^7.4.0",
120
+ "openai": "^7.5.0",
121
121
  "undici": "^7.29.0"
122
122
  },
123
123
  "lint-staged": {
@@ -126,8 +126,8 @@
126
126
  "devDependencies": {
127
127
  "gpt-tokenizer": "^3.4.0",
128
128
  "lint-staged": "^17.3.0",
129
- "release-it": "^21.0.1",
129
+ "release-it": "^21.0.2",
130
130
  "standard": "^17.1.2",
131
- "vitest": "^4.1.10"
131
+ "vitest": "^4.1.11"
132
132
  }
133
133
  }
package/src/cli/ask.js CHANGED
@@ -176,6 +176,12 @@ Examples:
176
176
  if (tokens.outputTokens) summary.push(`${tokens.outputTokens} out`)
177
177
  if (tokens.thinkingTokens) summary.push(`${tokens.thinkingTokens} think`)
178
178
  if (tokens.cost != null) summary.push(`$${tokens.cost.toFixed(4)}`)
179
+ if (tokens.speed) {
180
+ const served = tokens.servedSpeed
181
+ summary.push(served === tokens.speed
182
+ ? `${tokens.speed} lane`
183
+ : `${tokens.speed} lane → ${served ?? 'standard'}`)
184
+ }
179
185
  const ts = tokens.timestamps
180
186
  if (ts) {
181
187
  const toMs = (a, b) => {
package/src/cli/check.js CHANGED
@@ -1,7 +1,7 @@
1
1
  import { label, err, warn, ok } from './colors.js'
2
2
  import providers from '../lib/providers.js'
3
3
  import { validate, isValidTag } from '../lib/schema.js'
4
- import { providerSupportsSpeed } from '../../js/session/adapters/_speed.js'
4
+ import { adapters } from '../../js/session/adapters/index.js'
5
5
  import { getCuratedModels, loadDefaultEnv, catalogEntries, catalogValues } from '../lib/common.js'
6
6
 
7
7
  // --- Local validation ---
@@ -51,10 +51,12 @@ const checkLocal = (curated) => {
51
51
  warnings.push(`${key}: has thinkingEffortLevels but no defaultThinkingEffort`)
52
52
  }
53
53
 
54
- if (spec.speeds && !providerSupportsSpeed(keyProvider)) {
55
- errors.push(`${key}: declares speeds but provider '${keyProvider}' has no adapter support — calls on those lanes would fail at dispatch`)
56
- }
54
+ const lanes = adapters[keyProvider]?.speedLanes
57
55
  for (const [lane, overlay] of Object.entries(spec.speeds || {})) {
56
+ if (!lanes?.has(lane)) {
57
+ const detail = lanes ? `accepts: ${[...lanes].join(', ')}` : 'implements no speed lanes'
58
+ errors.push(`${key}: speeds.${lane} — provider '${keyProvider}' ${detail}; calls on that lane would fail at dispatch`)
59
+ }
58
60
  for (const priceField of ['inputPrice', 'outputPrice', 'thinkingPrice']) {
59
61
  const val = overlay[priceField]
60
62
  if (val != null && typeof val === 'object' && val.default == null) {
package/src/lib/index.js CHANGED
@@ -75,6 +75,17 @@ const resolvePrice = (price, inputTokens) => {
75
75
  // @internal — exported for unit tests only.
76
76
  export { resolvePrice as _resolvePriceForTests }
77
77
 
78
+ // A lane candidate containing `:` means the suffixes were written the
79
+ // other way round, which is otherwise reported as an unknown lane
80
+ // spelled `fast:none`.
81
+ const reversedSuffixHint = (base, candidate) => {
82
+ const colon = candidate.indexOf(':')
83
+ if (colon < 0) return ''
84
+ const speed = candidate.slice(0, colon)
85
+ const effort = candidate.slice(colon + 1)
86
+ return ` Suffix order is ':effort' then '@speed' — did you mean '${base}:${effort}@${speed}'?`
87
+ }
88
+
78
89
  const normalizeModelSpec = (resolvedModelId, modelSpec, providerConfig) => {
79
90
  const normalized = { ...modelSpec }
80
91
  if (!normalized.provider) {
@@ -357,9 +368,12 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
357
368
  if (aliasSpeed && !Object.hasOwn(modelSpec.speeds || {}, aliasSpeed)) {
358
369
  const available = Object.keys(modelSpec.speeds || {})
359
370
  const detail = available.length
360
- ? `Available: ${available.join(', ')}`
371
+ ? `Available: ${available.join(', ')}.`
361
372
  : 'It declares no speed lanes.'
362
- throw new Error(`Model '${resolvedModelId}' does not support speed lane '${aliasSpeed}'. ${detail}`)
373
+ throw new Error(
374
+ `Model '${resolvedModelId}' does not support speed lane '${aliasSpeed}'. ${detail}` +
375
+ reversedSuffixHint(resolvedModelId, aliasSpeed)
376
+ )
363
377
  }
364
378
 
365
379
  // Validate outputEffort alias against model capabilities
package/src/lib/schema.js CHANGED
@@ -1,14 +1,11 @@
1
1
  import { SPEED_OVERRIDABLE } from '../../js/session/adapters/_speed.js'
2
2
 
3
3
  const validateSpeeds = (speeds) => {
4
- const allowed = new Set(['wire', ...SPEED_OVERRIDABLE])
4
+ const allowed = new Set(SPEED_OVERRIDABLE)
5
5
  for (const [lane, overlay] of Object.entries(speeds)) {
6
6
  if (typeof overlay !== 'object' || overlay === null || Array.isArray(overlay)) {
7
7
  return `lane '${lane}' must be an object`
8
8
  }
9
- if (typeof overlay.wire !== 'string' || !overlay.wire) {
10
- return `lane '${lane}' must set a string 'wire' value`
11
- }
12
9
  for (const field of Object.keys(overlay)) {
13
10
  if (!allowed.has(field)) {
14
11
  return `lane '${lane}' may not override '${field}' (allowed: ${[...allowed].join(', ')})`
@@ -30,6 +27,9 @@ const fieldDefs = {
30
27
  inputPrice: { type: 'number', altType: 'object' },
31
28
  outputPrice: { type: 'number', altType: 'object' },
32
29
  thinkingPrice: { type: 'number', altType: 'object' },
30
+ cacheReadPrice: { type: 'number', altType: 'object' },
31
+ cacheWritePrice: { type: 'number', altType: 'object' },
32
+ cacheWrite1hPrice: { type: 'number', altType: 'object' },
33
33
  contextTokenLimit: { type: 'number' },
34
34
  outputTokenLimit: { type: 'number' },
35
35
  thinkingTokenLimit: { type: 'number' },