mohdel 0.119.0 → 0.120.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -101,6 +101,9 @@ mo ask anthropic/claude-sonnet-4-6 --stream "write a haiku about recursion"
101
101
  # With thinking effort
102
102
  mo ask anthropic/claude-opus-4-6 --effort high "prove P != NP"
103
103
 
104
+ # On a faster service lane, when the model sells one
105
+ mo ask anthropic/claude-opus-4-6@fast "triage this alert"
106
+
104
107
  # Speech → text from an audio file
105
108
  mo transcribe groq/whisper-large-v3-turbo meeting.mp3
106
109
  mo transcribe mistral/voxtral-mini-transcribe interview.wav --language fr
@@ -315,6 +318,8 @@ What each provider supports through mohdel's unified interface:
315
318
 
316
319
  Adapter capability ≠ model capability — whether a given model accepts images, tools, or thinking effort depends on the model spec in `curated.json`. The adapter passes through what the envelope supplies; the provider rejects unsupported combos.
317
320
 
321
+ **Service speeds** are the exception to that pass-through rule. Where a provider sells the same weights at several speeds (Anthropic's fast mode, and equivalents elsewhere), the lanes a model sells are declared in its `curated.json` entry under `speeds`, with their own prices and rate limits. Select one with `speed` or the `@lane` id suffix. There is no default lane and no fallback: an undeclared lane fails the call before it is sent, because a model that silently ignores an unsupported lane would otherwise be billed at the lane's rates for standard service. See [docs/CATALOG.md](docs/CATALOG.md#service-speeds).
322
+
318
323
  ## Local Development
319
324
 
320
325
  ```bash
@@ -88,6 +88,7 @@
88
88
  "description": "USD per 1M thinking/reasoning tokens (when the provider bills these separately)."
89
89
  },
90
90
  "cacheWritePrice": { "type": "number", "minimum": 0, "description": "USD per 1M tokens written to provider-side prompt cache." },
91
+ "cacheWrite1hPrice": { "type": "number", "minimum": 0, "description": "USD per 1M tokens written with a 1h TTL, when the provider prices that above the default write rate. Falls back to cacheWritePrice." },
91
92
  "cacheReadPrice": { "type": "number", "minimum": 0, "description": "USD per 1M tokens served from provider-side prompt cache." },
92
93
 
93
94
  "contextTokenLimit": { "type": "integer", "minimum": 1, "description": "Maximum total tokens (input + output)." },
@@ -107,6 +108,43 @@
107
108
  },
108
109
  "defaultThinkingEffort": { "type": "string", "description": "Effort level used when the envelope omits 'outputEffort'." },
109
110
 
111
+ "speeds": {
112
+ "type": "object",
113
+ "description": "Service speed lanes this model sells, keyed by lane name. A lane is a provider request parameter buying a different speed/price point for the same weights. There is no default lane: omitting 'speed' on the envelope sends no parameter. Only lanes declared here may be requested.",
114
+ "additionalProperties": {
115
+ "type": "object",
116
+ "required": ["wire"],
117
+ "additionalProperties": false,
118
+ "description": "Lane overlay. Fields named here override the base entry; unnamed fields fall through. Only prices and rate limits are overridable — anything that would change what the model is belongs in its own entry.",
119
+ "properties": {
120
+ "wire": { "type": "string", "description": "Provider-native value for the lane parameter (e.g. 'fast')." },
121
+ "inputPrice": {
122
+ "oneOf": [
123
+ { "type": "number", "minimum": 0 },
124
+ { "type": "object", "required": ["default"], "additionalProperties": { "type": "number" }, "properties": { "default": { "type": "number" } } }
125
+ ]
126
+ },
127
+ "outputPrice": {
128
+ "oneOf": [
129
+ { "type": "number", "minimum": 0 },
130
+ { "type": "object", "required": ["default"], "additionalProperties": { "type": "number" }, "properties": { "default": { "type": "number" } } }
131
+ ]
132
+ },
133
+ "thinkingPrice": {
134
+ "oneOf": [
135
+ { "type": "number", "minimum": 0 },
136
+ { "type": "object", "required": ["default"], "additionalProperties": { "type": "number" }, "properties": { "default": { "type": "number" } } }
137
+ ]
138
+ },
139
+ "cacheWritePrice": { "type": "number", "minimum": 0 },
140
+ "cacheWrite1hPrice": { "type": "number", "minimum": 0 },
141
+ "cacheReadPrice": { "type": "number", "minimum": 0 },
142
+ "rpmLimit": { "type": "integer", "minimum": 1 },
143
+ "tpmLimit": { "type": "integer", "minimum": 1 }
144
+ }
145
+ }
146
+ },
147
+
110
148
  "tags": {
111
149
  "type": "array",
112
150
  "items": { "type": "string", "pattern": "^[a-zA-Z][a-zA-Z0-9._-]{0,31}$" },
@@ -43,6 +43,13 @@
43
43
  * Thinking effort level. Valid keys are per-model — validated at
44
44
  * runtime against the curated entry's `thinkingEffortLevels`.
45
45
  * Default is the model's `defaultThinkingEffort`.
46
+ * @property {string} [speed]
47
+ * Service speed lane. Valid keys are per-model — validated at
48
+ * runtime against the curated entry's `speeds`. There is no
49
+ * default and no fallback: omitting the field sends no speed
50
+ * parameter, while naming a lane the entry does not declare
51
+ * (`SESSION_INVALID_SPEED`) or the provider's adapter cannot emit
52
+ * (`SESSION_SPEED_NOT_IMPLEMENTED`) fails the call before dispatch.
46
53
  *
47
54
  * @property {MediaRef[]} [images]
48
55
  * @property {MediaRef[]} [videos]
@@ -146,6 +153,7 @@ export const ENVELOPE_FIELDS = Object.freeze([
146
153
  'outputType',
147
154
  'outputStyle',
148
155
  'outputEffort',
156
+ 'speed',
149
157
  'images',
150
158
  'videos',
151
159
  'cache',
@@ -2,10 +2,10 @@
2
2
  * Model-id helpers.
3
3
  *
4
4
  * A mohdel model id is a single string of shape
5
- * `"<provider>/<bare>[:<effort>]"` — same on the wire and in-process.
6
- * See PROTOCOL §3. Nothing in mohdel ever holds the id in a split
7
- * object form; when the provider or bare part is needed, these
8
- * helpers return it as a substring.
5
+ * `"<provider>/<bare>[:<effort>][@<speed>]"` — same on the wire and
6
+ * in-process. See PROTOCOL §3. Nothing in mohdel ever holds the id in
7
+ * a split object form; when a part is needed, these helpers return it
8
+ * as a substring.
9
9
  *
10
10
  * Ids are validated at ingress by the gate
11
11
  * (`rust/thin-gate/src/protocol.rs::validate_ids`); these accessors
@@ -47,12 +47,8 @@ export function bareOf (model) {
47
47
  * @returns {string}
48
48
  */
49
49
  export function catalogKey (model) {
50
- const colon = model.lastIndexOf(':')
51
- const slash = model.indexOf('/')
52
- // Only treat `:` as an effort separator when it appears after the
53
- // provider slash (otherwise a model id without `/` that happens to
54
- // contain `:` would get the wrong thing stripped).
55
- return colon > slash ? model.slice(0, colon) : model
50
+ const base = beforeSuffix(model, SPEED_SIGIL)
51
+ return beforeSuffix(base, EFFORT_SIGIL)
56
52
  }
57
53
 
58
54
  /**
@@ -62,8 +58,46 @@ export function catalogKey (model) {
62
58
  * @returns {string | undefined}
63
59
  */
64
60
  export function effortOf (model) {
65
- const colon = model.lastIndexOf(':')
61
+ return afterSuffix(beforeSuffix(model, SPEED_SIGIL), EFFORT_SIGIL)
62
+ }
63
+
64
+ /**
65
+ * Speed-lane suffix, without the `@`, or `undefined` if absent.
66
+ *
67
+ * @param {string} model
68
+ * @returns {string | undefined}
69
+ */
70
+ export function speedOf (model) {
71
+ return afterSuffix(model, SPEED_SIGIL)
72
+ }
73
+
74
+ const EFFORT_SIGIL = ':'
75
+ const SPEED_SIGIL = '@'
76
+
77
+ /**
78
+ * Index of `sigil` when it separates a suffix, or `-1`. A sigil only
79
+ * separates when it falls after the provider slash, so a bare id that
80
+ * itself contains one is left whole.
81
+ *
82
+ * @param {string} model
83
+ * @param {string} sigil
84
+ * @returns {number}
85
+ */
86
+ function suffixIndex (model, sigil) {
66
87
  const slash = model.indexOf('/')
67
- if (colon <= slash) return undefined
68
- return model.slice(colon + 1)
88
+ if (slash < 0) return -1
89
+ const at = model.lastIndexOf(sigil)
90
+ return at > slash ? at : -1
91
+ }
92
+
93
+ /** @returns {string} */
94
+ function beforeSuffix (model, sigil) {
95
+ const at = suffixIndex(model, sigil)
96
+ return at < 0 ? model : model.slice(0, at)
97
+ }
98
+
99
+ /** @returns {string | undefined} */
100
+ function afterSuffix (model, sigil) {
101
+ const at = suffixIndex(model, sigil)
102
+ return at < 0 ? undefined : model.slice(at + 1)
69
103
  }
@@ -223,6 +223,7 @@ function toEnvelope ({ modelKey, configuration, prompt, options }) {
223
223
  if (options.outputType) envelope.outputType = options.outputType
224
224
  if (options.outputStyle) envelope.outputStyle = options.outputStyle
225
225
  if (options.outputEffort) envelope.outputEffort = options.outputEffort
226
+ if (options.speed) envelope.speed = options.speed
226
227
  if (options.images?.length) envelope.images = options.images
227
228
  if (options.videos?.length) envelope.videos = options.videos
228
229
  if (options.cache !== undefined) envelope.cache = options.cache
@@ -13,7 +13,6 @@
13
13
 
14
14
  import { STATUS_INCOMPLETE, WARNING_CANCELLED } from '#core/status.js'
15
15
  import { costFor } from './_pricing.js'
16
- import { catalogKey } from '#core/model-id.js'
17
16
 
18
17
  /**
19
18
  * @param {string} start hrtime-bigint-as-string at call entry
@@ -43,7 +42,7 @@ export function cancelledDone (start, first, envelope, output, inputTokens, outp
43
42
  ...(cacheWriteInputTokens > 0 && { cacheWriteInputTokens }),
44
43
  ...(cacheReadInputTokens > 0 && { cacheReadInputTokens }),
45
44
  cost: costFor(
46
- catalogKey(envelope.model),
45
+ envelope,
47
46
  { inputTokens, outputTokens, thinkingTokens: 0, cacheWriteInputTokens, cacheReadInputTokens }
48
47
  ),
49
48
  timestamps: { start, first: first ?? end, end },
@@ -12,7 +12,10 @@
12
12
 
13
13
  import envPaths from 'env-paths'
14
14
 
15
+ import { catalogKey } from '#core/model-id.js'
16
+
15
17
  import { createLazyJsonFileCache } from './_lazy_json_cache.js'
18
+ import { mergeSpeed } from './_speed.js'
16
19
 
17
20
  // `{ suffix: null }` mirrors `src/lib/common.js::CONFIG_DIR` so the
18
21
  // session subprocess reads the same `~/.config/mohdel/curated.json`
@@ -56,3 +59,16 @@ export function setCatalog (table) {
56
59
  export function getSpec (model) {
57
60
  return cache.get(model)
58
61
  }
62
+
63
+ /**
64
+ * Effective spec for a call: the catalog entry with any speed-lane
65
+ * overlay applied. Adapters resolve spec through this rather than
66
+ * `getSpec` so lane prices and quotas reach pricing and throttling
67
+ * without each adapter merging for itself.
68
+ *
69
+ * @param {import('#core/envelope.js').CallEnvelope} envelope
70
+ * @returns {any | undefined}
71
+ */
72
+ export function specFor (envelope) {
73
+ return mergeSpeed(getSpec(catalogKey(envelope.model)), envelope.speed)
74
+ }
@@ -19,6 +19,7 @@
19
19
  import { getSpec } from './_catalog.js'
20
20
  import { classifyProviderError } from './_errors.js'
21
21
  import { costFor } from './_pricing.js'
22
+ import { applySpeed } from './_speed.js'
22
23
  import { catalogKey, bareOf } from '#core/model-id.js'
23
24
  import {
24
25
  STATUS_COMPLETED,
@@ -287,7 +288,7 @@ function finalize ({ envelope, content, toolCalls, usage, finishReason, start, f
287
288
  thinkingTokens,
288
289
  ...(cachedInputTokens > 0 && { cacheReadInputTokens: cachedInputTokens }),
289
290
  cost: costFor(
290
- catalogKey(envelope.model),
291
+ envelope,
291
292
  {
292
293
  inputTokens,
293
294
  outputTokens: visibleOutputTokens,
@@ -370,6 +371,8 @@ function buildRequest (envelope, spec, config) {
370
371
  args[config.identifierField || 'user'] = envelope.identifier
371
372
  }
372
373
 
374
+ applySpeed(args, envelope, spec)
375
+
373
376
  return args
374
377
  }
375
378
 
@@ -53,6 +53,29 @@ function scrubKey (detail, key) {
53
53
  return detail.split(key).join(mask)
54
54
  }
55
55
 
56
+ const UNWRAP_DEPTH = 2
57
+
58
+ /**
59
+ * `@google/genai` sets `ApiError.message` to `JSON.stringify({error:
60
+ * {message, code, status}})`, and Gemini's own body inside it is
61
+ * itself JSON — so the sentence sits two envelopes deep.
62
+ * @param {string} str
63
+ * @returns {string}
64
+ */
65
+ function unwrapJsonMessage (str) {
66
+ let current = str
67
+ for (let i = 0; i < UNWRAP_DEPTH; i++) {
68
+ const trimmed = current.trim()
69
+ if (!trimmed.startsWith('{')) return current
70
+ let parsed
71
+ try { parsed = JSON.parse(trimmed) } catch { return current }
72
+ const inner = parsed?.error?.message ?? parsed?.message
73
+ if (typeof inner !== 'string' || !inner) return current
74
+ current = inner
75
+ }
76
+ return current
77
+ }
78
+
56
79
  /**
57
80
  * Extract a short human-readable detail from an SDK error. Trimmed
58
81
  * to `DETAIL_CAP` chars so a verbose provider body doesn't blow up
@@ -65,7 +88,7 @@ function extractDetail (err) {
65
88
  const nested = err.error?.message || err.response?.data?.error?.message
66
89
  const raw = nested || err.message
67
90
  if (!raw) return undefined
68
- const str = typeof raw === 'string' ? raw : JSON.stringify(raw)
91
+ const str = unwrapJsonMessage(typeof raw === 'string' ? raw : JSON.stringify(raw))
69
92
  return str.length > DETAIL_CAP ? str.slice(0, DETAIL_CAP) + '…' : str
70
93
  }
71
94
 
@@ -9,7 +9,7 @@
9
9
  * @module session/adapters/_pricing
10
10
  */
11
11
 
12
- import { getSpec, setCatalog } from './_catalog.js'
12
+ import { setCatalog, specFor } from './_catalog.js'
13
13
 
14
14
  /**
15
15
  * Pure cost computation from spec + usage.
@@ -104,12 +104,15 @@ function resolveTier (price, tokens) {
104
104
  }
105
105
 
106
106
  /**
107
- * @param {string} model Fully-qualified `<provider>/<model>`.
107
+ * Cost of a call, priced against the envelope's effective spec so an
108
+ * active speed lane bills at the lane's rates.
109
+ *
110
+ * @param {import('#core/envelope.js').CallEnvelope} envelope
108
111
  * @param {{inputTokens?: number, outputTokens?: number, thinkingTokens?: number}} usage
109
112
  * @returns {number}
110
113
  */
111
- export function costFor (model, usage) {
112
- return computeCost(getSpec(model), usage)
114
+ export function costFor (envelope, usage) {
115
+ return computeCost(specFor(envelope), usage)
113
116
  }
114
117
 
115
118
  /**
@@ -0,0 +1,119 @@
1
+ /**
2
+ * Service-speed lanes.
3
+ *
4
+ * A lane is a provider request parameter that buys a different
5
+ * speed/price point for the same weights (Anthropic `speed: "fast"`).
6
+ * Lanes are declared per catalog entry under `speeds`, keyed by lane
7
+ * name, each carrying the wire value plus any price and rate-limit
8
+ * fields that differ from the base entry.
9
+ *
10
+ * Two declarations must agree for a lane to be usable: the entry
11
+ * declares it (`spec.speeds`) and the provider's adapter can emit it
12
+ * (`SPEED_PARAMS`). `run.js` checks both before dispatch.
13
+ *
14
+ * @module session/adapters/_speed
15
+ */
16
+
17
+ import { providerOf } from '#core/model-id.js'
18
+
19
+ /**
20
+ * Providers whose adapter emits a lane parameter, mapped to the
21
+ * native parameter name. Absence from this table is the declaration
22
+ * that a provider has no lanes.
23
+ */
24
+ export const SPEED_PARAMS = Object.freeze({
25
+ anthropic: 'speed'
26
+ })
27
+
28
+ /** Spec fields a lane overlay may restate. */
29
+ export const SPEED_OVERRIDABLE = Object.freeze([
30
+ 'inputPrice',
31
+ 'outputPrice',
32
+ 'thinkingPrice',
33
+ 'cacheWritePrice',
34
+ 'cacheWrite1hPrice',
35
+ 'cacheReadPrice',
36
+ 'rpmLimit',
37
+ 'tpmLimit'
38
+ ])
39
+
40
+ /**
41
+ * @param {string} provider
42
+ * @returns {boolean}
43
+ */
44
+ export function providerSupportsSpeed (provider) {
45
+ return Object.hasOwn(SPEED_PARAMS, provider)
46
+ }
47
+
48
+ /**
49
+ * @param {string} provider
50
+ * @returns {string | undefined}
51
+ */
52
+ export function speedParamFor (provider) {
53
+ return SPEED_PARAMS[provider]
54
+ }
55
+
56
+ /**
57
+ * Whether `spec` declares `speed`. Uses `hasOwn` rather than
58
+ * truthiness so an overlay that only carries `wire` still counts.
59
+ *
60
+ * @param {any} spec
61
+ * @param {string} speed
62
+ * @returns {boolean}
63
+ */
64
+ export function hasSpeed (spec, speed) {
65
+ return !!spec?.speeds && Object.hasOwn(spec.speeds, speed)
66
+ }
67
+
68
+ /**
69
+ * @param {any} spec
70
+ * @returns {string[]}
71
+ */
72
+ export function speedNames (spec) {
73
+ return spec?.speeds ? Object.keys(spec.speeds) : []
74
+ }
75
+
76
+ /**
77
+ * Spec with `speed`'s overlay applied. Unnamed fields fall through to
78
+ * the base entry.
79
+ *
80
+ * @param {any} spec
81
+ * @param {string} [speed]
82
+ * @returns {any}
83
+ * @throws when `speed` is set and `spec` does not declare it
84
+ */
85
+ export function mergeSpeed (spec, speed) {
86
+ if (!speed) return spec
87
+ if (!hasSpeed(spec, speed)) {
88
+ throw new Error(`model does not declare speed lane '${speed}'`)
89
+ }
90
+ const overlay = spec.speeds[speed]
91
+ const merged = { ...spec }
92
+ for (const field of SPEED_OVERRIDABLE) {
93
+ if (Object.hasOwn(overlay, field)) merged[field] = overlay[field]
94
+ }
95
+ return merged
96
+ }
97
+
98
+ /**
99
+ * Set the provider-native lane parameter on an outbound request.
100
+ * No-op when the envelope carries no lane.
101
+ *
102
+ * @param {Record<string, any>} request
103
+ * @param {import('#core/envelope.js').CallEnvelope} envelope
104
+ * @param {any} spec
105
+ * @throws when the envelope carries a lane the provider cannot emit
106
+ * or the spec does not declare
107
+ */
108
+ export function applySpeed (request, envelope, spec) {
109
+ if (!envelope.speed) return
110
+ const provider = providerOf(envelope.model)
111
+ const param = speedParamFor(provider)
112
+ if (!param) {
113
+ throw new Error(`provider '${provider}' does not implement speed lanes`)
114
+ }
115
+ if (!hasSpeed(spec, envelope.speed)) {
116
+ throw new Error(`model does not declare speed lane '${envelope.speed}'`)
117
+ }
118
+ request[param] = spec.speeds[envelope.speed].wire
119
+ }
@@ -29,6 +29,7 @@ import { classifyProviderError } from './_errors.js'
29
29
  import { loadImages } from './_images.js'
30
30
  import { isTrustedMedia } from './_media.js'
31
31
  import { costFor } from './_pricing.js'
32
+ import { applySpeed } from './_speed.js'
32
33
  import { catalogKey, bareOf } from '#core/model-id.js'
33
34
  import {
34
35
  toAnthropicTools,
@@ -261,7 +262,7 @@ export async function * anthropic (envelope, deps = {}) {
261
262
  ...(cacheWrite1hTokens > 0 && { cacheWrite1hInputTokens: cacheWrite1hTokens }),
262
263
  ...(cacheReadTokens > 0 && { cacheReadInputTokens: cacheReadTokens }),
263
264
  cost: costFor(
264
- catalogKey(envelope.model),
265
+ envelope,
265
266
  {
266
267
  inputTokens,
267
268
  outputTokens: messageOutputTokens,
@@ -352,6 +353,7 @@ function buildRequest (envelope, conversation, system, conversationCacheTtl = nu
352
353
  }
353
354
  }
354
355
 
356
+ applySpeed(request, envelope, spec)
355
357
  applyCacheBreakpoints(request, conversationCacheTtl)
356
358
 
357
359
  return request
@@ -31,6 +31,7 @@ import { loadImages } from './_images.js'
31
31
  import { isTrustedMedia } from './_media.js'
32
32
  import { loadVideos } from './_videos.js'
33
33
  import { costFor } from './_pricing.js'
34
+ import { applySpeed } from './_speed.js'
34
35
  import { catalogKey, bareOf } from '#core/model-id.js'
35
36
  import {
36
37
  toGeminiTools,
@@ -208,7 +209,7 @@ export async function * gemini (envelope, deps = {}) {
208
209
  thinkingTokens,
209
210
  ...(cacheReadTokens > 0 && { cacheReadInputTokens: cacheReadTokens }),
210
211
  cost: costFor(
211
- catalogKey(envelope.model),
212
+ envelope,
212
213
  { inputTokens: regularInputTokens, outputTokens, thinkingTokens, cacheReadInputTokens: cacheReadTokens }
213
214
  ),
214
215
  timestamps: { start, first: first ?? end, end }
@@ -271,6 +272,7 @@ function buildRequest (envelope, contents, systemInstruction) {
271
272
  contents
272
273
  }
273
274
  if (Object.keys(config).length > 0) request.config = config
275
+ applySpeed(request, envelope, spec)
274
276
  return request
275
277
  }
276
278
 
@@ -30,6 +30,7 @@ import { classifyProviderError } from './_errors.js'
30
30
  import { loadImages } from './_images.js'
31
31
  import { isTrustedMedia } from './_media.js'
32
32
  import { costFor } from './_pricing.js'
33
+ import { applySpeed } from './_speed.js'
33
34
  import { catalogKey, providerOf, bareOf } from '#core/model-id.js'
34
35
  import {
35
36
  toOpenAITools,
@@ -201,7 +202,7 @@ export async function * openai (envelope, deps = {}) {
201
202
  ...(cacheWriteTokens > 0 && { cacheWriteInputTokens: cacheWriteTokens }),
202
203
  ...(cachedInputTokens > 0 && { cacheReadInputTokens: cachedInputTokens }),
203
204
  cost: costFor(
204
- catalogKey(envelope.model),
205
+ envelope,
205
206
  {
206
207
  inputTokens: regularInputTokens,
207
208
  outputTokens: messageOutputTokens,
@@ -287,6 +288,8 @@ function buildRequest (envelope, input, instructions) {
287
288
  }
288
289
  }
289
290
 
291
+ applySpeed(request, envelope, spec)
292
+
290
293
  return request
291
294
  }
292
295
 
package/js/session/run.js CHANGED
@@ -26,7 +26,8 @@ import { getAdapter } from './adapters/index.js'
26
26
  import { isImageProvider } from './adapters/image/index.js'
27
27
  import { getSpec } from './adapters/_catalog.js'
28
28
  import { getProviderLimits } from './adapters/_providers.js'
29
- import { providerOf, catalogKey, effortOf } from '#core/model-id.js'
29
+ import { hasSpeed, mergeSpeed, providerSupportsSpeed, speedNames } from './adapters/_speed.js'
30
+ import { providerOf, catalogKey, effortOf, speedOf } from '#core/model-id.js'
30
31
  import * as defaultCooldown from './_cooldown.js'
31
32
  import * as defaultLimiter from './_rate_limiter.js'
32
33
  import { withIdleHeartbeat, MIN_IDLE_HEARTBEAT_MS } from './_idle_heartbeat.js'
@@ -66,15 +67,13 @@ export async function * run (envelope, {
66
67
  sleep = defaultSleep,
67
68
  signal
68
69
  } = {}) {
69
- // Honor the `model:effort` shortcut on the wire (mirrors the
70
- // factory-side `mohdel().use('model:effort')` convenience). If
71
- // the envelope's `model` field ends in `:<effort>` and the base
72
- // resolves to a known spec, split the suffix into
73
- // `envelope.outputEffort`. Explicit `outputEffort` wins when both
74
- // are set (suffix is a shortcut, not an override).
75
- const effortNorm = normalizeModelEffort(envelope, resolveSpec)
76
- if (effortNorm.error) { yield effortNorm.error; return }
77
- envelope = effortNorm.envelope
70
+ // Honor the `model:effort@speed` shortcuts on the wire (mirrors the
71
+ // factory-side `mohdel().use('model:effort')` convenience). Explicit
72
+ // envelope fields win when both are set (a suffix is a shortcut, not
73
+ // an override).
74
+ const norm = normalizeModelId(envelope, resolveSpec)
75
+ if (norm.error) { yield norm.error; return }
76
+ envelope = norm.envelope
78
77
 
79
78
  const provider = providerOf(envelope.model)
80
79
  const span = openSpan(envelope)
@@ -85,6 +84,7 @@ export async function * run (envelope, {
85
84
  provider,
86
85
  model: envelope.model,
87
86
  effort: envelope.outputEffort ?? 'default',
87
+ speed: envelope.speed ?? null,
88
88
  outputBudget: envelope.outputBudget ?? null,
89
89
  tools: envelope.tools?.length || 0,
90
90
  images: envelope.images?.length || 0
@@ -116,11 +116,8 @@ export async function * run (envelope, {
116
116
  // Catalog is authoritative: every callable model must have a
117
117
  // spec. Without one we'd silently run the provider call with
118
118
  // defaults (no rate-limits, no budget clamps, cost=0), masking
119
- // misconfiguration in the layer that pushed the catalog. Effort
120
- // suffix is stripped for the lookup — catalog entries are keyed
121
- // by the bare `<provider>/<bare>` id, not per-effort variants.
122
- const key = catalogKey(envelope.model)
123
- const spec = resolveSpec(key)
119
+ // misconfiguration in the layer that pushed the catalog.
120
+ const { key, spec } = norm
124
121
  if (!spec) {
125
122
  const detail = `Unknown model '${key}' — not in catalog`
126
123
  const err = errorEvent(detail, 'SESSION_UNKNOWN_MODEL')
@@ -130,6 +127,17 @@ export async function * run (envelope, {
130
127
  return
131
128
  }
132
129
 
130
+ if (envelope.speed) {
131
+ const speedErr = speedError(key, envelope.speed, spec, provider)
132
+ if (speedErr) {
133
+ log.warn({ provider, speed: envelope.speed }, '[mohdel:answer] unusable speed lane')
134
+ endSpanError(span, new Error(speedErr.error.message))
135
+ yield speedErr
136
+ return
137
+ }
138
+ }
139
+ const effective = mergeSpeed(spec, envelope.speed)
140
+
133
141
  const coolErr = cooldown.coolingDownError(provider)
134
142
  if (coolErr) {
135
143
  log.debug({ provider, detail: coolErr.detail }, '[mohdel:cooldown] fast-fail')
@@ -140,9 +148,14 @@ export async function * run (envelope, {
140
148
  }
141
149
 
142
150
  const providerCfg = resolveProviderLimits(provider) || {}
143
- const rpmLimit = spec?.rpmLimit ?? providerCfg.rpmLimit
144
- const tpmLimit = spec?.tpmLimit ?? providerCfg.tpmLimit
145
- const bucketKey = (spec?.rateLimitScope === 'model') ? key : provider
151
+ const rpmLimit = effective?.rpmLimit ?? providerCfg.rpmLimit
152
+ const tpmLimit = effective?.tpmLimit ?? providerCfg.tpmLimit
153
+ // An active lane is its own capacity pool, so it gets its own bucket
154
+ // whatever `rateLimitScope` says — sharing one would let standard
155
+ // traffic throttle the lane being paid for.
156
+ const bucketKey = envelope.speed
157
+ ? `${key}@${envelope.speed}`
158
+ : (spec?.rateLimitScope === 'model' ? key : provider)
146
159
 
147
160
  // `0` is a killswitch ("deny all"), not "unset"; `undefined`/`null`
148
161
  // means no limit configured for that dimension. Gate on nullability
@@ -220,8 +233,12 @@ export async function * run (envelope, {
220
233
  }
221
234
  // Surface on AnswerResult so hosts that pass the whole
222
235
  // result upstream pick it up without needing a separate
223
- // wire field.
224
- if (ev.result) ev.result.maxInterFrameMs = maxInterFrameMs
236
+ // wire field. `speed` rides along because lane prices differ,
237
+ // so cost is only meaningful attributed per (model, lane).
238
+ if (ev.result) {
239
+ ev.result.maxInterFrameMs = maxInterFrameMs
240
+ if (envelope.speed) ev.result.speed = envelope.speed
241
+ }
225
242
  finalizeSpanOk(span, ev.result, sawDelta, maxInterFrameMs)
226
243
  log.debug(summarizeDone(ev.result, startedAt), '[mohdel:answer] done')
227
244
  } else if (ev.type === 'error') {
@@ -267,53 +284,108 @@ export async function * run (envelope, {
267
284
  }
268
285
 
269
286
  /**
270
- * Split an optional `:effort` suffix from `envelope.model`. If the
271
- * base resolves to a known spec, rewrites `envelope.model` and sets
272
- * `envelope.outputEffort` (unless already set). Emits a typed error
273
- * when the suffix is present and the spec rejects it.
287
+ * Split the optional `:effort` and `@speed` suffixes from
288
+ * `envelope.model`. If the base resolves to a known spec, rewrites
289
+ * `envelope.model` and sets `envelope.outputEffort` / `envelope.speed`
290
+ * (unless already set). Emits a typed error when an effort suffix is
291
+ * present and the spec rejects it; lane validity is checked in `run`
292
+ * so that an explicitly-set `envelope.speed` goes through the same
293
+ * guard as the suffix form.
294
+ *
295
+ * Also carries out the catalog lookup, so the key the spec was found
296
+ * under is the one the rest of the call uses.
274
297
  *
275
298
  * @param {import('#core/envelope.js').CallEnvelope} envelope
276
299
  * @param {(key: string) => any} resolveSpec
277
300
  * @returns {{
278
301
  * envelope: import('#core/envelope.js').CallEnvelope,
302
+ * key: string,
303
+ * spec?: any,
279
304
  * error?: import('#core/events.js').ErrorEvent
280
305
  * }}
281
306
  */
282
- function normalizeModelEffort (envelope, resolveSpec) {
283
- const candidate = effortOf(envelope.model)
284
- if (!candidate) return { envelope }
307
+ function normalizeModelId (envelope, resolveSpec) {
308
+ // A bare id that itself contains `:` or `@` is a catalog key in its
309
+ // own right; resolving the whole string first stops it being split
310
+ // into a base plus a suffix that was never meant as one.
311
+ const whole = resolveSpec(envelope.model)
312
+ if (whole) return { envelope, key: envelope.model, spec: whole }
285
313
 
314
+ const effort = effortOf(envelope.model)
315
+ const speed = speedOf(envelope.model)
286
316
  const base = catalogKey(envelope.model)
287
317
  const baseSpec = resolveSpec(base)
288
- if (!baseSpec) return { envelope } // base not known — let full string fall through to not-found
318
+ const unresolved = { envelope, key: base, spec: baseSpec }
319
+ if (effort === undefined && speed === undefined) return unresolved
320
+ if (!baseSpec) return unresolved
289
321
 
290
- // Explicit outputEffort wins; still strip the suffix so spans/logs see the canonical id.
291
- if (envelope.outputEffort) {
292
- return { envelope: { ...envelope, model: base } }
322
+ const next = { ...envelope, model: base }
323
+ if (speed !== undefined && !envelope.speed) next.speed = speed
324
+
325
+ if (effort === undefined || envelope.outputEffort) {
326
+ return { envelope: next, key: base, spec: baseSpec }
293
327
  }
294
328
 
295
329
  if (!baseSpec.thinkingEffortLevels) {
296
330
  return {
297
- envelope,
331
+ ...unresolved,
298
332
  error: errorEvent(
299
- `Model '${base}' does not support output effort (no thinkingEffortLevels). Cannot use ':${candidate}' suffix.`,
333
+ `Model '${base}' does not support output effort (no thinkingEffortLevels). Cannot use ':${effort}' suffix.`,
300
334
  'SESSION_INVALID_OUTPUT_EFFORT'
301
335
  )
302
336
  }
303
337
  }
304
- if (candidate !== 'none' && !baseSpec.thinkingEffortLevels[candidate]) {
338
+ if (effort !== 'none' && !baseSpec.thinkingEffortLevels[effort]) {
305
339
  return {
306
- envelope,
340
+ ...unresolved,
307
341
  error: errorEvent(
308
- `Model '${base}' does not support output effort level '${candidate}'. Available: ${Object.keys(baseSpec.thinkingEffortLevels).join(', ')}`,
342
+ `Model '${base}' does not support output effort level '${effort}'. Available: ${Object.keys(baseSpec.thinkingEffortLevels).join(', ')}`,
309
343
  'SESSION_INVALID_OUTPUT_EFFORT'
310
344
  )
311
345
  }
312
346
  }
313
347
 
314
- return {
315
- envelope: { ...envelope, model: base, outputEffort: candidate }
348
+ next.outputEffort = effort
349
+ return { envelope: next, key: base, spec: baseSpec }
350
+ }
351
+
352
+ /**
353
+ * The two lane guards, in caller-then-internal order. Returns an
354
+ * error event, or `undefined` when the lane is usable.
355
+ *
356
+ * They are deliberately separate. A lane the entry does not declare is
357
+ * the caller asking for something this model does not sell. A lane the
358
+ * adapter cannot emit is the catalog and the adapter disagreeing —
359
+ * left to run, the call would silently take the standard lane and bill
360
+ * at the overlay's rates.
361
+ *
362
+ * @param {string} key
363
+ * @param {string} speed
364
+ * @param {any} spec
365
+ * @param {string} provider
366
+ * @returns {import('#core/events.js').ErrorEvent | undefined}
367
+ */
368
+ function speedError (key, speed, spec, provider) {
369
+ if (!hasSpeed(spec, speed)) {
370
+ const available = speedNames(spec)
371
+ const detail = available.length
372
+ ? `Available: ${available.join(', ')}`
373
+ : 'It declares no speed lanes.'
374
+ const hint = speed.includes(':')
375
+ ? " Suffix order is ':effort' then '@speed'."
376
+ : ''
377
+ return errorEvent(
378
+ `Model '${key}' does not support speed lane '${speed}'. ${detail}${hint}`,
379
+ 'SESSION_INVALID_SPEED'
380
+ )
381
+ }
382
+ if (!providerSupportsSpeed(provider)) {
383
+ return errorEvent(
384
+ `Provider '${provider}' does not implement speed lanes, but '${key}' declares '${speed}'.`,
385
+ 'SESSION_SPEED_NOT_IMPLEMENTED'
386
+ )
316
387
  }
388
+ return undefined
317
389
  }
318
390
 
319
391
  /**
@@ -332,6 +404,7 @@ function openSpan (envelope) {
332
404
  }
333
405
  if (envelope.outputBudget) attrs['gen_ai.request.max_tokens'] = envelope.outputBudget
334
406
  if (envelope.outputEffort) attrs['mohdel.output_effort'] = envelope.outputEffort
407
+ if (envelope.speed) attrs['mohdel.speed'] = envelope.speed
335
408
  return startSpan('mohdel.session.answer', attrs, parent)
336
409
  }
337
410
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mohdel",
3
- "version": "0.119.0",
3
+ "version": "0.120.0",
4
4
  "license": "MIT",
5
5
  "author": {
6
6
  "name": "Christophe Le Bars",
@@ -108,7 +108,7 @@
108
108
  "@opentelemetry/exporter-trace-otlp-grpc": "^0.221.0",
109
109
  "@opentelemetry/sdk-node": "^0.221.0",
110
110
  "chalk": "^6.0.0",
111
- "mohdel-thin-gate-linux-x64-gnu": "0.119.0"
111
+ "mohdel-thin-gate-linux-x64-gnu": "0.120.0"
112
112
  },
113
113
  "dependencies": {
114
114
  "@anthropic-ai/sdk": "^0.115.0",
package/src/cli/check.js CHANGED
@@ -1,6 +1,7 @@
1
1
  import { label, err, warn, ok } from './colors.js'
2
2
  import providers from '../lib/providers.js'
3
3
  import { validate, isValidTag } from '../lib/schema.js'
4
+ import { providerSupportsSpeed } from '../../js/session/adapters/_speed.js'
4
5
  import { getCuratedModels, loadDefaultEnv, catalogEntries, catalogValues } from '../lib/common.js'
5
6
 
6
7
  // --- Local validation ---
@@ -50,6 +51,22 @@ const checkLocal = (curated) => {
50
51
  warnings.push(`${key}: has thinkingEffortLevels but no defaultThinkingEffort`)
51
52
  }
52
53
 
54
+ if (spec.speeds && !providerSupportsSpeed(keyProvider)) {
55
+ errors.push(`${key}: declares speeds but provider '${keyProvider}' has no adapter support — calls on those lanes would fail at dispatch`)
56
+ }
57
+ for (const [lane, overlay] of Object.entries(spec.speeds || {})) {
58
+ for (const priceField of ['inputPrice', 'outputPrice', 'thinkingPrice']) {
59
+ const val = overlay[priceField]
60
+ if (val != null && typeof val === 'object' && val.default == null) {
61
+ errors.push(`${key}: speeds.${lane}.${priceField} is tiered but missing 'default' key`)
62
+ }
63
+ }
64
+ const priced = ['inputPrice', 'outputPrice'].some(f => overlay[f] != null)
65
+ if (!priced) {
66
+ warnings.push(`${key}: speeds.${lane} restates no prices — the lane will bill at base rates`)
67
+ }
68
+ }
69
+
53
70
  if (Array.isArray(spec.tags)) {
54
71
  for (const t of spec.tags) {
55
72
  if (!isValidTag(t)) warnings.push(`${key}: invalid tag "${t}" — must match /^[a-zA-Z][a-zA-Z0-9._-]{0,31}$/`)
package/src/lib/index.js CHANGED
@@ -268,8 +268,35 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
268
268
  // below. If `base` doesn't resolve, fall through — the full
269
269
  // `modelId` (with colon) gets the normal lookup + "not
270
270
  // found" error path.
271
+ // A curated id that itself contains `:` or `@` is a catalog
272
+ // key in its own right, so an exact hit wins over splitting.
273
+ // Fallback specs are excluded from that test on purpose:
274
+ // providers that synthesize them resolve any string, which
275
+ // would disable suffix parsing for them entirely.
276
+ const exactId = libraryMode ? modelId : expandModelAliasSync(modelId)
277
+ const isCuratedId = !!catalog[exactId]
278
+
279
+ // `@speed` is parsed off first so the two suffixes are read
280
+ // in canonical order (`base:effort@speed`).
281
+ let aliasSpeed
282
+ const atIdx = isCuratedId ? -1 : modelId.lastIndexOf('@')
283
+ if (atIdx > 0) {
284
+ const candidate = modelId.slice(atIdx + 1)
285
+ const base = modelId.slice(0, atIdx)
286
+ // The base may still carry `:effort`, which is stripped
287
+ // separately below — probe without it so `x:high@fast`
288
+ // resolves the same as `x@fast`.
289
+ const colon = base.lastIndexOf(':')
290
+ const probe = colon > 0 ? base.slice(0, colon) : base
291
+ const probeResolved = libraryMode ? probe : expandModelAliasSync(probe)
292
+ if (catalog[probeResolved] || createFallbackModelSpec(probeResolved)) {
293
+ aliasSpeed = candidate
294
+ modelId = base
295
+ }
296
+ }
297
+
271
298
  let aliasOutputEffort
272
- const colonIdx = modelId.lastIndexOf(':')
299
+ const colonIdx = isCuratedId ? -1 : modelId.lastIndexOf(':')
273
300
  if (colonIdx > 0) {
274
301
  const candidate = modelId.slice(colonIdx + 1)
275
302
  const base = modelId.slice(0, colonIdx)
@@ -327,6 +354,14 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
327
354
  }
328
355
  modelSpec = normalizeModelSpec(resolvedModelId, modelSpec, providerConfig)
329
356
 
357
+ if (aliasSpeed && !Object.hasOwn(modelSpec.speeds || {}, aliasSpeed)) {
358
+ const available = Object.keys(modelSpec.speeds || {})
359
+ const detail = available.length
360
+ ? `Available: ${available.join(', ')}`
361
+ : 'It declares no speed lanes.'
362
+ throw new Error(`Model '${resolvedModelId}' does not support speed lane '${aliasSpeed}'. ${detail}`)
363
+ }
364
+
330
365
  // Validate outputEffort alias against model capabilities
331
366
  if (aliasOutputEffort) {
332
367
  if (!modelSpec.thinkingEffortLevels) {
@@ -337,7 +372,7 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
337
372
  }
338
373
  }
339
374
 
340
- return createModelProxy(resolvedModelId, modelSpec, handlers, aliasOutputEffort, sdkCache, rateLimiter, providersConfig, cooldown, configurations, resolveProviderLimits)
375
+ return createModelProxy(resolvedModelId, modelSpec, handlers, aliasOutputEffort, aliasSpeed, sdkCache, rateLimiter, providersConfig, cooldown, configurations, resolveProviderLimits)
341
376
  }
342
377
  }
343
378
 
@@ -383,7 +418,7 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
383
418
  })
384
419
  }
385
420
 
386
- const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffort, sdkCache, rateLimiter, providersConfig, cooldown, externalConfigurations, resolveProviderLimits) => {
421
+ const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffort, aliasSpeed, sdkCache, rateLimiter, providersConfig, cooldown, externalConfigurations, resolveProviderLimits) => {
387
422
  // modelSpec is the full metadata object for resolvedModelId
388
423
  let runtimePromise = null
389
424
 
@@ -445,6 +480,9 @@ const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffor
445
480
  if (aliasOutputEffort && !options.outputEffort) {
446
481
  options.outputEffort = aliasOutputEffort
447
482
  }
483
+ if (aliasSpeed && !options.speed) {
484
+ options.speed = aliasSpeed
485
+ }
448
486
 
449
487
  // Verbosity tier — captured from handlers (set by the mohdel factory). Used
450
488
  // to gate per-call log lines. Read once per call so the closure doesn't have
package/src/lib/schema.js CHANGED
@@ -1,3 +1,23 @@
1
+ import { SPEED_OVERRIDABLE } from '../../js/session/adapters/_speed.js'
2
+
3
+ const validateSpeeds = (speeds) => {
4
+ const allowed = new Set(['wire', ...SPEED_OVERRIDABLE])
5
+ for (const [lane, overlay] of Object.entries(speeds)) {
6
+ if (typeof overlay !== 'object' || overlay === null || Array.isArray(overlay)) {
7
+ return `lane '${lane}' must be an object`
8
+ }
9
+ if (typeof overlay.wire !== 'string' || !overlay.wire) {
10
+ return `lane '${lane}' must set a string 'wire' value`
11
+ }
12
+ for (const field of Object.keys(overlay)) {
13
+ if (!allowed.has(field)) {
14
+ return `lane '${lane}' may not override '${field}' (allowed: ${[...allowed].join(', ')})`
15
+ }
16
+ }
17
+ }
18
+ return null
19
+ }
20
+
1
21
  const fieldDefs = {
2
22
  model: { type: 'string', required: true },
3
23
  provider: { type: 'string' },
@@ -15,6 +35,7 @@ const fieldDefs = {
15
35
  thinkingTokenLimit: { type: 'number' },
16
36
  thinkingEffortLevels: { type: 'object', nullable: true, default: null },
17
37
  defaultThinkingEffort: { type: 'string' },
38
+ speeds: { type: 'object', validate: validateSpeeds },
18
39
  tags: { type: 'array', itemType: 'string', default: [] },
19
40
  aliases: { type: 'array', itemType: 'string', default: [] },
20
41
  replaces: { type: 'array', itemType: 'string', default: [] },