mohdel 0.119.0 → 0.120.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -0
- package/config/curated.schema.json +38 -0
- package/js/core/envelope.js +8 -0
- package/js/core/model-id.js +47 -13
- package/js/factory/bridge.js +1 -0
- package/js/session/adapters/_cancelled.js +1 -2
- package/js/session/adapters/_catalog.js +16 -0
- package/js/session/adapters/_chat_completions.js +4 -1
- package/js/session/adapters/_errors.js +24 -1
- package/js/session/adapters/_pricing.js +7 -4
- package/js/session/adapters/_speed.js +119 -0
- package/js/session/adapters/anthropic.js +3 -1
- package/js/session/adapters/gemini.js +3 -1
- package/js/session/adapters/openai.js +4 -1
- package/js/session/run.js +111 -38
- package/package.json +2 -2
- package/src/cli/check.js +17 -0
- package/src/lib/index.js +41 -3
- package/src/lib/schema.js +21 -0
package/README.md
CHANGED
|
@@ -101,6 +101,9 @@ mo ask anthropic/claude-sonnet-4-6 --stream "write a haiku about recursion"
|
|
|
101
101
|
# With thinking effort
|
|
102
102
|
mo ask anthropic/claude-opus-4-6 --effort high "prove P != NP"
|
|
103
103
|
|
|
104
|
+
# On a faster service lane, when the model sells one
|
|
105
|
+
mo ask anthropic/claude-opus-4-6@fast "triage this alert"
|
|
106
|
+
|
|
104
107
|
# Speech → text from an audio file
|
|
105
108
|
mo transcribe groq/whisper-large-v3-turbo meeting.mp3
|
|
106
109
|
mo transcribe mistral/voxtral-mini-transcribe interview.wav --language fr
|
|
@@ -315,6 +318,8 @@ What each provider supports through mohdel's unified interface:
|
|
|
315
318
|
|
|
316
319
|
Adapter capability ≠ model capability — whether a given model accepts images, tools, or thinking effort depends on the model spec in `curated.json`. The adapter passes through what the envelope supplies; the provider rejects unsupported combos.
|
|
317
320
|
|
|
321
|
+
**Service speeds** are the exception to that pass-through rule. Where a provider sells the same weights at several speeds (Anthropic's fast mode, and equivalents elsewhere), the lanes a model sells are declared in its `curated.json` entry under `speeds`, with their own prices and rate limits. Select one with `speed` or the `@lane` id suffix. There is no default lane and no fallback: an undeclared lane fails the call before it is sent, because a model that silently ignores an unsupported lane would otherwise be billed at the lane's rates for standard service. See [docs/CATALOG.md](docs/CATALOG.md#service-speeds).
|
|
322
|
+
|
|
318
323
|
## Local Development
|
|
319
324
|
|
|
320
325
|
```bash
|
|
@@ -88,6 +88,7 @@
|
|
|
88
88
|
"description": "USD per 1M thinking/reasoning tokens (when the provider bills these separately)."
|
|
89
89
|
},
|
|
90
90
|
"cacheWritePrice": { "type": "number", "minimum": 0, "description": "USD per 1M tokens written to provider-side prompt cache." },
|
|
91
|
+
"cacheWrite1hPrice": { "type": "number", "minimum": 0, "description": "USD per 1M tokens written with a 1h TTL, when the provider prices that above the default write rate. Falls back to cacheWritePrice." },
|
|
91
92
|
"cacheReadPrice": { "type": "number", "minimum": 0, "description": "USD per 1M tokens served from provider-side prompt cache." },
|
|
92
93
|
|
|
93
94
|
"contextTokenLimit": { "type": "integer", "minimum": 1, "description": "Maximum total tokens (input + output)." },
|
|
@@ -107,6 +108,43 @@
|
|
|
107
108
|
},
|
|
108
109
|
"defaultThinkingEffort": { "type": "string", "description": "Effort level used when the envelope omits 'outputEffort'." },
|
|
109
110
|
|
|
111
|
+
"speeds": {
|
|
112
|
+
"type": "object",
|
|
113
|
+
"description": "Service speed lanes this model sells, keyed by lane name. A lane is a provider request parameter buying a different speed/price point for the same weights. There is no default lane: omitting 'speed' on the envelope sends no parameter. Only lanes declared here may be requested.",
|
|
114
|
+
"additionalProperties": {
|
|
115
|
+
"type": "object",
|
|
116
|
+
"required": ["wire"],
|
|
117
|
+
"additionalProperties": false,
|
|
118
|
+
"description": "Lane overlay. Fields named here override the base entry; unnamed fields fall through. Only prices and rate limits are overridable — anything that would change what the model is belongs in its own entry.",
|
|
119
|
+
"properties": {
|
|
120
|
+
"wire": { "type": "string", "description": "Provider-native value for the lane parameter (e.g. 'fast')." },
|
|
121
|
+
"inputPrice": {
|
|
122
|
+
"oneOf": [
|
|
123
|
+
{ "type": "number", "minimum": 0 },
|
|
124
|
+
{ "type": "object", "required": ["default"], "additionalProperties": { "type": "number" }, "properties": { "default": { "type": "number" } } }
|
|
125
|
+
]
|
|
126
|
+
},
|
|
127
|
+
"outputPrice": {
|
|
128
|
+
"oneOf": [
|
|
129
|
+
{ "type": "number", "minimum": 0 },
|
|
130
|
+
{ "type": "object", "required": ["default"], "additionalProperties": { "type": "number" }, "properties": { "default": { "type": "number" } } }
|
|
131
|
+
]
|
|
132
|
+
},
|
|
133
|
+
"thinkingPrice": {
|
|
134
|
+
"oneOf": [
|
|
135
|
+
{ "type": "number", "minimum": 0 },
|
|
136
|
+
{ "type": "object", "required": ["default"], "additionalProperties": { "type": "number" }, "properties": { "default": { "type": "number" } } }
|
|
137
|
+
]
|
|
138
|
+
},
|
|
139
|
+
"cacheWritePrice": { "type": "number", "minimum": 0 },
|
|
140
|
+
"cacheWrite1hPrice": { "type": "number", "minimum": 0 },
|
|
141
|
+
"cacheReadPrice": { "type": "number", "minimum": 0 },
|
|
142
|
+
"rpmLimit": { "type": "integer", "minimum": 1 },
|
|
143
|
+
"tpmLimit": { "type": "integer", "minimum": 1 }
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
},
|
|
147
|
+
|
|
110
148
|
"tags": {
|
|
111
149
|
"type": "array",
|
|
112
150
|
"items": { "type": "string", "pattern": "^[a-zA-Z][a-zA-Z0-9._-]{0,31}$" },
|
package/js/core/envelope.js
CHANGED
|
@@ -43,6 +43,13 @@
|
|
|
43
43
|
* Thinking effort level. Valid keys are per-model — validated at
|
|
44
44
|
* runtime against the curated entry's `thinkingEffortLevels`.
|
|
45
45
|
* Default is the model's `defaultThinkingEffort`.
|
|
46
|
+
* @property {string} [speed]
|
|
47
|
+
* Service speed lane. Valid keys are per-model — validated at
|
|
48
|
+
* runtime against the curated entry's `speeds`. There is no
|
|
49
|
+
* default and no fallback: omitting the field sends no speed
|
|
50
|
+
* parameter, while naming a lane the entry does not declare
|
|
51
|
+
* (`SESSION_INVALID_SPEED`) or the provider's adapter cannot emit
|
|
52
|
+
* (`SESSION_SPEED_NOT_IMPLEMENTED`) fails the call before dispatch.
|
|
46
53
|
*
|
|
47
54
|
* @property {MediaRef[]} [images]
|
|
48
55
|
* @property {MediaRef[]} [videos]
|
|
@@ -146,6 +153,7 @@ export const ENVELOPE_FIELDS = Object.freeze([
|
|
|
146
153
|
'outputType',
|
|
147
154
|
'outputStyle',
|
|
148
155
|
'outputEffort',
|
|
156
|
+
'speed',
|
|
149
157
|
'images',
|
|
150
158
|
'videos',
|
|
151
159
|
'cache',
|
package/js/core/model-id.js
CHANGED
|
@@ -2,10 +2,10 @@
|
|
|
2
2
|
* Model-id helpers.
|
|
3
3
|
*
|
|
4
4
|
* A mohdel model id is a single string of shape
|
|
5
|
-
* `"<provider>/<bare>[:<effort>]"` — same on the wire and
|
|
6
|
-
* See PROTOCOL §3. Nothing in mohdel ever holds the id in
|
|
7
|
-
* object form; when
|
|
8
|
-
*
|
|
5
|
+
* `"<provider>/<bare>[:<effort>][@<speed>]"` — same on the wire and
|
|
6
|
+
* in-process. See PROTOCOL §3. Nothing in mohdel ever holds the id in
|
|
7
|
+
* a split object form; when a part is needed, these helpers return it
|
|
8
|
+
* as a substring.
|
|
9
9
|
*
|
|
10
10
|
* Ids are validated at ingress by the gate
|
|
11
11
|
* (`rust/thin-gate/src/protocol.rs::validate_ids`); these accessors
|
|
@@ -47,12 +47,8 @@ export function bareOf (model) {
|
|
|
47
47
|
* @returns {string}
|
|
48
48
|
*/
|
|
49
49
|
export function catalogKey (model) {
|
|
50
|
-
const
|
|
51
|
-
|
|
52
|
-
// Only treat `:` as an effort separator when it appears after the
|
|
53
|
-
// provider slash (otherwise a model id without `/` that happens to
|
|
54
|
-
// contain `:` would get the wrong thing stripped).
|
|
55
|
-
return colon > slash ? model.slice(0, colon) : model
|
|
50
|
+
const base = beforeSuffix(model, SPEED_SIGIL)
|
|
51
|
+
return beforeSuffix(base, EFFORT_SIGIL)
|
|
56
52
|
}
|
|
57
53
|
|
|
58
54
|
/**
|
|
@@ -62,8 +58,46 @@ export function catalogKey (model) {
|
|
|
62
58
|
* @returns {string | undefined}
|
|
63
59
|
*/
|
|
64
60
|
export function effortOf (model) {
|
|
65
|
-
|
|
61
|
+
return afterSuffix(beforeSuffix(model, SPEED_SIGIL), EFFORT_SIGIL)
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Speed-lane suffix, without the `@`, or `undefined` if absent.
|
|
66
|
+
*
|
|
67
|
+
* @param {string} model
|
|
68
|
+
* @returns {string | undefined}
|
|
69
|
+
*/
|
|
70
|
+
export function speedOf (model) {
|
|
71
|
+
return afterSuffix(model, SPEED_SIGIL)
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
const EFFORT_SIGIL = ':'
|
|
75
|
+
const SPEED_SIGIL = '@'
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Index of `sigil` when it separates a suffix, or `-1`. A sigil only
|
|
79
|
+
* separates when it falls after the provider slash, so a bare id that
|
|
80
|
+
* itself contains one is left whole.
|
|
81
|
+
*
|
|
82
|
+
* @param {string} model
|
|
83
|
+
* @param {string} sigil
|
|
84
|
+
* @returns {number}
|
|
85
|
+
*/
|
|
86
|
+
function suffixIndex (model, sigil) {
|
|
66
87
|
const slash = model.indexOf('/')
|
|
67
|
-
if (
|
|
68
|
-
|
|
88
|
+
if (slash < 0) return -1
|
|
89
|
+
const at = model.lastIndexOf(sigil)
|
|
90
|
+
return at > slash ? at : -1
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/** @returns {string} */
|
|
94
|
+
function beforeSuffix (model, sigil) {
|
|
95
|
+
const at = suffixIndex(model, sigil)
|
|
96
|
+
return at < 0 ? model : model.slice(0, at)
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/** @returns {string | undefined} */
|
|
100
|
+
function afterSuffix (model, sigil) {
|
|
101
|
+
const at = suffixIndex(model, sigil)
|
|
102
|
+
return at < 0 ? undefined : model.slice(at + 1)
|
|
69
103
|
}
|
package/js/factory/bridge.js
CHANGED
|
@@ -223,6 +223,7 @@ function toEnvelope ({ modelKey, configuration, prompt, options }) {
|
|
|
223
223
|
if (options.outputType) envelope.outputType = options.outputType
|
|
224
224
|
if (options.outputStyle) envelope.outputStyle = options.outputStyle
|
|
225
225
|
if (options.outputEffort) envelope.outputEffort = options.outputEffort
|
|
226
|
+
if (options.speed) envelope.speed = options.speed
|
|
226
227
|
if (options.images?.length) envelope.images = options.images
|
|
227
228
|
if (options.videos?.length) envelope.videos = options.videos
|
|
228
229
|
if (options.cache !== undefined) envelope.cache = options.cache
|
|
@@ -13,7 +13,6 @@
|
|
|
13
13
|
|
|
14
14
|
import { STATUS_INCOMPLETE, WARNING_CANCELLED } from '#core/status.js'
|
|
15
15
|
import { costFor } from './_pricing.js'
|
|
16
|
-
import { catalogKey } from '#core/model-id.js'
|
|
17
16
|
|
|
18
17
|
/**
|
|
19
18
|
* @param {string} start hrtime-bigint-as-string at call entry
|
|
@@ -43,7 +42,7 @@ export function cancelledDone (start, first, envelope, output, inputTokens, outp
|
|
|
43
42
|
...(cacheWriteInputTokens > 0 && { cacheWriteInputTokens }),
|
|
44
43
|
...(cacheReadInputTokens > 0 && { cacheReadInputTokens }),
|
|
45
44
|
cost: costFor(
|
|
46
|
-
|
|
45
|
+
envelope,
|
|
47
46
|
{ inputTokens, outputTokens, thinkingTokens: 0, cacheWriteInputTokens, cacheReadInputTokens }
|
|
48
47
|
),
|
|
49
48
|
timestamps: { start, first: first ?? end, end },
|
|
@@ -12,7 +12,10 @@
|
|
|
12
12
|
|
|
13
13
|
import envPaths from 'env-paths'
|
|
14
14
|
|
|
15
|
+
import { catalogKey } from '#core/model-id.js'
|
|
16
|
+
|
|
15
17
|
import { createLazyJsonFileCache } from './_lazy_json_cache.js'
|
|
18
|
+
import { mergeSpeed } from './_speed.js'
|
|
16
19
|
|
|
17
20
|
// `{ suffix: null }` mirrors `src/lib/common.js::CONFIG_DIR` so the
|
|
18
21
|
// session subprocess reads the same `~/.config/mohdel/curated.json`
|
|
@@ -56,3 +59,16 @@ export function setCatalog (table) {
|
|
|
56
59
|
export function getSpec (model) {
|
|
57
60
|
return cache.get(model)
|
|
58
61
|
}
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Effective spec for a call: the catalog entry with any speed-lane
|
|
65
|
+
* overlay applied. Adapters resolve spec through this rather than
|
|
66
|
+
* `getSpec` so lane prices and quotas reach pricing and throttling
|
|
67
|
+
* without each adapter merging for itself.
|
|
68
|
+
*
|
|
69
|
+
* @param {import('#core/envelope.js').CallEnvelope} envelope
|
|
70
|
+
* @returns {any | undefined}
|
|
71
|
+
*/
|
|
72
|
+
export function specFor (envelope) {
|
|
73
|
+
return mergeSpeed(getSpec(catalogKey(envelope.model)), envelope.speed)
|
|
74
|
+
}
|
|
@@ -19,6 +19,7 @@
|
|
|
19
19
|
import { getSpec } from './_catalog.js'
|
|
20
20
|
import { classifyProviderError } from './_errors.js'
|
|
21
21
|
import { costFor } from './_pricing.js'
|
|
22
|
+
import { applySpeed } from './_speed.js'
|
|
22
23
|
import { catalogKey, bareOf } from '#core/model-id.js'
|
|
23
24
|
import {
|
|
24
25
|
STATUS_COMPLETED,
|
|
@@ -287,7 +288,7 @@ function finalize ({ envelope, content, toolCalls, usage, finishReason, start, f
|
|
|
287
288
|
thinkingTokens,
|
|
288
289
|
...(cachedInputTokens > 0 && { cacheReadInputTokens: cachedInputTokens }),
|
|
289
290
|
cost: costFor(
|
|
290
|
-
|
|
291
|
+
envelope,
|
|
291
292
|
{
|
|
292
293
|
inputTokens,
|
|
293
294
|
outputTokens: visibleOutputTokens,
|
|
@@ -370,6 +371,8 @@ function buildRequest (envelope, spec, config) {
|
|
|
370
371
|
args[config.identifierField || 'user'] = envelope.identifier
|
|
371
372
|
}
|
|
372
373
|
|
|
374
|
+
applySpeed(args, envelope, spec)
|
|
375
|
+
|
|
373
376
|
return args
|
|
374
377
|
}
|
|
375
378
|
|
|
@@ -53,6 +53,29 @@ function scrubKey (detail, key) {
|
|
|
53
53
|
return detail.split(key).join(mask)
|
|
54
54
|
}
|
|
55
55
|
|
|
56
|
+
const UNWRAP_DEPTH = 2
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* `@google/genai` sets `ApiError.message` to `JSON.stringify({error:
|
|
60
|
+
* {message, code, status}})`, and Gemini's own body inside it is
|
|
61
|
+
* itself JSON — so the sentence sits two envelopes deep.
|
|
62
|
+
* @param {string} str
|
|
63
|
+
* @returns {string}
|
|
64
|
+
*/
|
|
65
|
+
function unwrapJsonMessage (str) {
|
|
66
|
+
let current = str
|
|
67
|
+
for (let i = 0; i < UNWRAP_DEPTH; i++) {
|
|
68
|
+
const trimmed = current.trim()
|
|
69
|
+
if (!trimmed.startsWith('{')) return current
|
|
70
|
+
let parsed
|
|
71
|
+
try { parsed = JSON.parse(trimmed) } catch { return current }
|
|
72
|
+
const inner = parsed?.error?.message ?? parsed?.message
|
|
73
|
+
if (typeof inner !== 'string' || !inner) return current
|
|
74
|
+
current = inner
|
|
75
|
+
}
|
|
76
|
+
return current
|
|
77
|
+
}
|
|
78
|
+
|
|
56
79
|
/**
|
|
57
80
|
* Extract a short human-readable detail from an SDK error. Trimmed
|
|
58
81
|
* to `DETAIL_CAP` chars so a verbose provider body doesn't blow up
|
|
@@ -65,7 +88,7 @@ function extractDetail (err) {
|
|
|
65
88
|
const nested = err.error?.message || err.response?.data?.error?.message
|
|
66
89
|
const raw = nested || err.message
|
|
67
90
|
if (!raw) return undefined
|
|
68
|
-
const str = typeof raw === 'string' ? raw : JSON.stringify(raw)
|
|
91
|
+
const str = unwrapJsonMessage(typeof raw === 'string' ? raw : JSON.stringify(raw))
|
|
69
92
|
return str.length > DETAIL_CAP ? str.slice(0, DETAIL_CAP) + '…' : str
|
|
70
93
|
}
|
|
71
94
|
|
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
* @module session/adapters/_pricing
|
|
10
10
|
*/
|
|
11
11
|
|
|
12
|
-
import {
|
|
12
|
+
import { setCatalog, specFor } from './_catalog.js'
|
|
13
13
|
|
|
14
14
|
/**
|
|
15
15
|
* Pure cost computation from spec + usage.
|
|
@@ -104,12 +104,15 @@ function resolveTier (price, tokens) {
|
|
|
104
104
|
}
|
|
105
105
|
|
|
106
106
|
/**
|
|
107
|
-
*
|
|
107
|
+
* Cost of a call, priced against the envelope's effective spec so an
|
|
108
|
+
* active speed lane bills at the lane's rates.
|
|
109
|
+
*
|
|
110
|
+
* @param {import('#core/envelope.js').CallEnvelope} envelope
|
|
108
111
|
* @param {{inputTokens?: number, outputTokens?: number, thinkingTokens?: number}} usage
|
|
109
112
|
* @returns {number}
|
|
110
113
|
*/
|
|
111
|
-
export function costFor (
|
|
112
|
-
return computeCost(
|
|
114
|
+
export function costFor (envelope, usage) {
|
|
115
|
+
return computeCost(specFor(envelope), usage)
|
|
113
116
|
}
|
|
114
117
|
|
|
115
118
|
/**
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Service-speed lanes.
|
|
3
|
+
*
|
|
4
|
+
* A lane is a provider request parameter that buys a different
|
|
5
|
+
* speed/price point for the same weights (Anthropic `speed: "fast"`).
|
|
6
|
+
* Lanes are declared per catalog entry under `speeds`, keyed by lane
|
|
7
|
+
* name, each carrying the wire value plus any price and rate-limit
|
|
8
|
+
* fields that differ from the base entry.
|
|
9
|
+
*
|
|
10
|
+
* Two declarations must agree for a lane to be usable: the entry
|
|
11
|
+
* declares it (`spec.speeds`) and the provider's adapter can emit it
|
|
12
|
+
* (`SPEED_PARAMS`). `run.js` checks both before dispatch.
|
|
13
|
+
*
|
|
14
|
+
* @module session/adapters/_speed
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import { providerOf } from '#core/model-id.js'
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* Providers whose adapter emits a lane parameter, mapped to the
|
|
21
|
+
* native parameter name. Absence from this table is the declaration
|
|
22
|
+
* that a provider has no lanes.
|
|
23
|
+
*/
|
|
24
|
+
export const SPEED_PARAMS = Object.freeze({
|
|
25
|
+
anthropic: 'speed'
|
|
26
|
+
})
|
|
27
|
+
|
|
28
|
+
/** Spec fields a lane overlay may restate. */
|
|
29
|
+
export const SPEED_OVERRIDABLE = Object.freeze([
|
|
30
|
+
'inputPrice',
|
|
31
|
+
'outputPrice',
|
|
32
|
+
'thinkingPrice',
|
|
33
|
+
'cacheWritePrice',
|
|
34
|
+
'cacheWrite1hPrice',
|
|
35
|
+
'cacheReadPrice',
|
|
36
|
+
'rpmLimit',
|
|
37
|
+
'tpmLimit'
|
|
38
|
+
])
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* @param {string} provider
|
|
42
|
+
* @returns {boolean}
|
|
43
|
+
*/
|
|
44
|
+
export function providerSupportsSpeed (provider) {
|
|
45
|
+
return Object.hasOwn(SPEED_PARAMS, provider)
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* @param {string} provider
|
|
50
|
+
* @returns {string | undefined}
|
|
51
|
+
*/
|
|
52
|
+
export function speedParamFor (provider) {
|
|
53
|
+
return SPEED_PARAMS[provider]
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Whether `spec` declares `speed`. Uses `hasOwn` rather than
|
|
58
|
+
* truthiness so an overlay that only carries `wire` still counts.
|
|
59
|
+
*
|
|
60
|
+
* @param {any} spec
|
|
61
|
+
* @param {string} speed
|
|
62
|
+
* @returns {boolean}
|
|
63
|
+
*/
|
|
64
|
+
export function hasSpeed (spec, speed) {
|
|
65
|
+
return !!spec?.speeds && Object.hasOwn(spec.speeds, speed)
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* @param {any} spec
|
|
70
|
+
* @returns {string[]}
|
|
71
|
+
*/
|
|
72
|
+
export function speedNames (spec) {
|
|
73
|
+
return spec?.speeds ? Object.keys(spec.speeds) : []
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* Spec with `speed`'s overlay applied. Unnamed fields fall through to
|
|
78
|
+
* the base entry.
|
|
79
|
+
*
|
|
80
|
+
* @param {any} spec
|
|
81
|
+
* @param {string} [speed]
|
|
82
|
+
* @returns {any}
|
|
83
|
+
* @throws when `speed` is set and `spec` does not declare it
|
|
84
|
+
*/
|
|
85
|
+
export function mergeSpeed (spec, speed) {
|
|
86
|
+
if (!speed) return spec
|
|
87
|
+
if (!hasSpeed(spec, speed)) {
|
|
88
|
+
throw new Error(`model does not declare speed lane '${speed}'`)
|
|
89
|
+
}
|
|
90
|
+
const overlay = spec.speeds[speed]
|
|
91
|
+
const merged = { ...spec }
|
|
92
|
+
for (const field of SPEED_OVERRIDABLE) {
|
|
93
|
+
if (Object.hasOwn(overlay, field)) merged[field] = overlay[field]
|
|
94
|
+
}
|
|
95
|
+
return merged
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* Set the provider-native lane parameter on an outbound request.
|
|
100
|
+
* No-op when the envelope carries no lane.
|
|
101
|
+
*
|
|
102
|
+
* @param {Record<string, any>} request
|
|
103
|
+
* @param {import('#core/envelope.js').CallEnvelope} envelope
|
|
104
|
+
* @param {any} spec
|
|
105
|
+
* @throws when the envelope carries a lane the provider cannot emit
|
|
106
|
+
* or the spec does not declare
|
|
107
|
+
*/
|
|
108
|
+
export function applySpeed (request, envelope, spec) {
|
|
109
|
+
if (!envelope.speed) return
|
|
110
|
+
const provider = providerOf(envelope.model)
|
|
111
|
+
const param = speedParamFor(provider)
|
|
112
|
+
if (!param) {
|
|
113
|
+
throw new Error(`provider '${provider}' does not implement speed lanes`)
|
|
114
|
+
}
|
|
115
|
+
if (!hasSpeed(spec, envelope.speed)) {
|
|
116
|
+
throw new Error(`model does not declare speed lane '${envelope.speed}'`)
|
|
117
|
+
}
|
|
118
|
+
request[param] = spec.speeds[envelope.speed].wire
|
|
119
|
+
}
|
|
@@ -29,6 +29,7 @@ import { classifyProviderError } from './_errors.js'
|
|
|
29
29
|
import { loadImages } from './_images.js'
|
|
30
30
|
import { isTrustedMedia } from './_media.js'
|
|
31
31
|
import { costFor } from './_pricing.js'
|
|
32
|
+
import { applySpeed } from './_speed.js'
|
|
32
33
|
import { catalogKey, bareOf } from '#core/model-id.js'
|
|
33
34
|
import {
|
|
34
35
|
toAnthropicTools,
|
|
@@ -261,7 +262,7 @@ export async function * anthropic (envelope, deps = {}) {
|
|
|
261
262
|
...(cacheWrite1hTokens > 0 && { cacheWrite1hInputTokens: cacheWrite1hTokens }),
|
|
262
263
|
...(cacheReadTokens > 0 && { cacheReadInputTokens: cacheReadTokens }),
|
|
263
264
|
cost: costFor(
|
|
264
|
-
|
|
265
|
+
envelope,
|
|
265
266
|
{
|
|
266
267
|
inputTokens,
|
|
267
268
|
outputTokens: messageOutputTokens,
|
|
@@ -352,6 +353,7 @@ function buildRequest (envelope, conversation, system, conversationCacheTtl = nu
|
|
|
352
353
|
}
|
|
353
354
|
}
|
|
354
355
|
|
|
356
|
+
applySpeed(request, envelope, spec)
|
|
355
357
|
applyCacheBreakpoints(request, conversationCacheTtl)
|
|
356
358
|
|
|
357
359
|
return request
|
|
@@ -31,6 +31,7 @@ import { loadImages } from './_images.js'
|
|
|
31
31
|
import { isTrustedMedia } from './_media.js'
|
|
32
32
|
import { loadVideos } from './_videos.js'
|
|
33
33
|
import { costFor } from './_pricing.js'
|
|
34
|
+
import { applySpeed } from './_speed.js'
|
|
34
35
|
import { catalogKey, bareOf } from '#core/model-id.js'
|
|
35
36
|
import {
|
|
36
37
|
toGeminiTools,
|
|
@@ -208,7 +209,7 @@ export async function * gemini (envelope, deps = {}) {
|
|
|
208
209
|
thinkingTokens,
|
|
209
210
|
...(cacheReadTokens > 0 && { cacheReadInputTokens: cacheReadTokens }),
|
|
210
211
|
cost: costFor(
|
|
211
|
-
|
|
212
|
+
envelope,
|
|
212
213
|
{ inputTokens: regularInputTokens, outputTokens, thinkingTokens, cacheReadInputTokens: cacheReadTokens }
|
|
213
214
|
),
|
|
214
215
|
timestamps: { start, first: first ?? end, end }
|
|
@@ -271,6 +272,7 @@ function buildRequest (envelope, contents, systemInstruction) {
|
|
|
271
272
|
contents
|
|
272
273
|
}
|
|
273
274
|
if (Object.keys(config).length > 0) request.config = config
|
|
275
|
+
applySpeed(request, envelope, spec)
|
|
274
276
|
return request
|
|
275
277
|
}
|
|
276
278
|
|
|
@@ -30,6 +30,7 @@ import { classifyProviderError } from './_errors.js'
|
|
|
30
30
|
import { loadImages } from './_images.js'
|
|
31
31
|
import { isTrustedMedia } from './_media.js'
|
|
32
32
|
import { costFor } from './_pricing.js'
|
|
33
|
+
import { applySpeed } from './_speed.js'
|
|
33
34
|
import { catalogKey, providerOf, bareOf } from '#core/model-id.js'
|
|
34
35
|
import {
|
|
35
36
|
toOpenAITools,
|
|
@@ -201,7 +202,7 @@ export async function * openai (envelope, deps = {}) {
|
|
|
201
202
|
...(cacheWriteTokens > 0 && { cacheWriteInputTokens: cacheWriteTokens }),
|
|
202
203
|
...(cachedInputTokens > 0 && { cacheReadInputTokens: cachedInputTokens }),
|
|
203
204
|
cost: costFor(
|
|
204
|
-
|
|
205
|
+
envelope,
|
|
205
206
|
{
|
|
206
207
|
inputTokens: regularInputTokens,
|
|
207
208
|
outputTokens: messageOutputTokens,
|
|
@@ -287,6 +288,8 @@ function buildRequest (envelope, input, instructions) {
|
|
|
287
288
|
}
|
|
288
289
|
}
|
|
289
290
|
|
|
291
|
+
applySpeed(request, envelope, spec)
|
|
292
|
+
|
|
290
293
|
return request
|
|
291
294
|
}
|
|
292
295
|
|
package/js/session/run.js
CHANGED
|
@@ -26,7 +26,8 @@ import { getAdapter } from './adapters/index.js'
|
|
|
26
26
|
import { isImageProvider } from './adapters/image/index.js'
|
|
27
27
|
import { getSpec } from './adapters/_catalog.js'
|
|
28
28
|
import { getProviderLimits } from './adapters/_providers.js'
|
|
29
|
-
import {
|
|
29
|
+
import { hasSpeed, mergeSpeed, providerSupportsSpeed, speedNames } from './adapters/_speed.js'
|
|
30
|
+
import { providerOf, catalogKey, effortOf, speedOf } from '#core/model-id.js'
|
|
30
31
|
import * as defaultCooldown from './_cooldown.js'
|
|
31
32
|
import * as defaultLimiter from './_rate_limiter.js'
|
|
32
33
|
import { withIdleHeartbeat, MIN_IDLE_HEARTBEAT_MS } from './_idle_heartbeat.js'
|
|
@@ -66,15 +67,13 @@ export async function * run (envelope, {
|
|
|
66
67
|
sleep = defaultSleep,
|
|
67
68
|
signal
|
|
68
69
|
} = {}) {
|
|
69
|
-
// Honor the `model:effort`
|
|
70
|
-
// factory-side `mohdel().use('model:effort')` convenience).
|
|
71
|
-
//
|
|
72
|
-
//
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
if (effortNorm.error) { yield effortNorm.error; return }
|
|
77
|
-
envelope = effortNorm.envelope
|
|
70
|
+
// Honor the `model:effort@speed` shortcuts on the wire (mirrors the
|
|
71
|
+
// factory-side `mohdel().use('model:effort')` convenience). Explicit
|
|
72
|
+
// envelope fields win when both are set (a suffix is a shortcut, not
|
|
73
|
+
// an override).
|
|
74
|
+
const norm = normalizeModelId(envelope, resolveSpec)
|
|
75
|
+
if (norm.error) { yield norm.error; return }
|
|
76
|
+
envelope = norm.envelope
|
|
78
77
|
|
|
79
78
|
const provider = providerOf(envelope.model)
|
|
80
79
|
const span = openSpan(envelope)
|
|
@@ -85,6 +84,7 @@ export async function * run (envelope, {
|
|
|
85
84
|
provider,
|
|
86
85
|
model: envelope.model,
|
|
87
86
|
effort: envelope.outputEffort ?? 'default',
|
|
87
|
+
speed: envelope.speed ?? null,
|
|
88
88
|
outputBudget: envelope.outputBudget ?? null,
|
|
89
89
|
tools: envelope.tools?.length || 0,
|
|
90
90
|
images: envelope.images?.length || 0
|
|
@@ -116,11 +116,8 @@ export async function * run (envelope, {
|
|
|
116
116
|
// Catalog is authoritative: every callable model must have a
|
|
117
117
|
// spec. Without one we'd silently run the provider call with
|
|
118
118
|
// defaults (no rate-limits, no budget clamps, cost=0), masking
|
|
119
|
-
// misconfiguration in the layer that pushed the catalog.
|
|
120
|
-
|
|
121
|
-
// by the bare `<provider>/<bare>` id, not per-effort variants.
|
|
122
|
-
const key = catalogKey(envelope.model)
|
|
123
|
-
const spec = resolveSpec(key)
|
|
119
|
+
// misconfiguration in the layer that pushed the catalog.
|
|
120
|
+
const { key, spec } = norm
|
|
124
121
|
if (!spec) {
|
|
125
122
|
const detail = `Unknown model '${key}' — not in catalog`
|
|
126
123
|
const err = errorEvent(detail, 'SESSION_UNKNOWN_MODEL')
|
|
@@ -130,6 +127,17 @@ export async function * run (envelope, {
|
|
|
130
127
|
return
|
|
131
128
|
}
|
|
132
129
|
|
|
130
|
+
if (envelope.speed) {
|
|
131
|
+
const speedErr = speedError(key, envelope.speed, spec, provider)
|
|
132
|
+
if (speedErr) {
|
|
133
|
+
log.warn({ provider, speed: envelope.speed }, '[mohdel:answer] unusable speed lane')
|
|
134
|
+
endSpanError(span, new Error(speedErr.error.message))
|
|
135
|
+
yield speedErr
|
|
136
|
+
return
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
const effective = mergeSpeed(spec, envelope.speed)
|
|
140
|
+
|
|
133
141
|
const coolErr = cooldown.coolingDownError(provider)
|
|
134
142
|
if (coolErr) {
|
|
135
143
|
log.debug({ provider, detail: coolErr.detail }, '[mohdel:cooldown] fast-fail')
|
|
@@ -140,9 +148,14 @@ export async function * run (envelope, {
|
|
|
140
148
|
}
|
|
141
149
|
|
|
142
150
|
const providerCfg = resolveProviderLimits(provider) || {}
|
|
143
|
-
const rpmLimit =
|
|
144
|
-
const tpmLimit =
|
|
145
|
-
|
|
151
|
+
const rpmLimit = effective?.rpmLimit ?? providerCfg.rpmLimit
|
|
152
|
+
const tpmLimit = effective?.tpmLimit ?? providerCfg.tpmLimit
|
|
153
|
+
// An active lane is its own capacity pool, so it gets its own bucket
|
|
154
|
+
// whatever `rateLimitScope` says — sharing one would let standard
|
|
155
|
+
// traffic throttle the lane being paid for.
|
|
156
|
+
const bucketKey = envelope.speed
|
|
157
|
+
? `${key}@${envelope.speed}`
|
|
158
|
+
: (spec?.rateLimitScope === 'model' ? key : provider)
|
|
146
159
|
|
|
147
160
|
// `0` is a killswitch ("deny all"), not "unset"; `undefined`/`null`
|
|
148
161
|
// means no limit configured for that dimension. Gate on nullability
|
|
@@ -220,8 +233,12 @@ export async function * run (envelope, {
|
|
|
220
233
|
}
|
|
221
234
|
// Surface on AnswerResult so hosts that pass the whole
|
|
222
235
|
// result upstream pick it up without needing a separate
|
|
223
|
-
// wire field.
|
|
224
|
-
|
|
236
|
+
// wire field. `speed` rides along because lane prices differ,
|
|
237
|
+
// so cost is only meaningful attributed per (model, lane).
|
|
238
|
+
if (ev.result) {
|
|
239
|
+
ev.result.maxInterFrameMs = maxInterFrameMs
|
|
240
|
+
if (envelope.speed) ev.result.speed = envelope.speed
|
|
241
|
+
}
|
|
225
242
|
finalizeSpanOk(span, ev.result, sawDelta, maxInterFrameMs)
|
|
226
243
|
log.debug(summarizeDone(ev.result, startedAt), '[mohdel:answer] done')
|
|
227
244
|
} else if (ev.type === 'error') {
|
|
@@ -267,53 +284,108 @@ export async function * run (envelope, {
|
|
|
267
284
|
}
|
|
268
285
|
|
|
269
286
|
/**
|
|
270
|
-
* Split
|
|
271
|
-
* base resolves to a known spec, rewrites
|
|
272
|
-
* `envelope.
|
|
273
|
-
*
|
|
287
|
+
* Split the optional `:effort` and `@speed` suffixes from
|
|
288
|
+
* `envelope.model`. If the base resolves to a known spec, rewrites
|
|
289
|
+
* `envelope.model` and sets `envelope.outputEffort` / `envelope.speed`
|
|
290
|
+
* (unless already set). Emits a typed error when an effort suffix is
|
|
291
|
+
* present and the spec rejects it; lane validity is checked in `run`
|
|
292
|
+
* so that an explicitly-set `envelope.speed` goes through the same
|
|
293
|
+
* guard as the suffix form.
|
|
294
|
+
*
|
|
295
|
+
* Also carries out the catalog lookup, so the key the spec was found
|
|
296
|
+
* under is the one the rest of the call uses.
|
|
274
297
|
*
|
|
275
298
|
* @param {import('#core/envelope.js').CallEnvelope} envelope
|
|
276
299
|
* @param {(key: string) => any} resolveSpec
|
|
277
300
|
* @returns {{
|
|
278
301
|
* envelope: import('#core/envelope.js').CallEnvelope,
|
|
302
|
+
* key: string,
|
|
303
|
+
* spec?: any,
|
|
279
304
|
* error?: import('#core/events.js').ErrorEvent
|
|
280
305
|
* }}
|
|
281
306
|
*/
|
|
282
|
-
function
|
|
283
|
-
|
|
284
|
-
|
|
307
|
+
function normalizeModelId (envelope, resolveSpec) {
|
|
308
|
+
// A bare id that itself contains `:` or `@` is a catalog key in its
|
|
309
|
+
// own right; resolving the whole string first stops it being split
|
|
310
|
+
// into a base plus a suffix that was never meant as one.
|
|
311
|
+
const whole = resolveSpec(envelope.model)
|
|
312
|
+
if (whole) return { envelope, key: envelope.model, spec: whole }
|
|
285
313
|
|
|
314
|
+
const effort = effortOf(envelope.model)
|
|
315
|
+
const speed = speedOf(envelope.model)
|
|
286
316
|
const base = catalogKey(envelope.model)
|
|
287
317
|
const baseSpec = resolveSpec(base)
|
|
288
|
-
|
|
318
|
+
const unresolved = { envelope, key: base, spec: baseSpec }
|
|
319
|
+
if (effort === undefined && speed === undefined) return unresolved
|
|
320
|
+
if (!baseSpec) return unresolved
|
|
289
321
|
|
|
290
|
-
|
|
291
|
-
if (envelope.
|
|
292
|
-
|
|
322
|
+
const next = { ...envelope, model: base }
|
|
323
|
+
if (speed !== undefined && !envelope.speed) next.speed = speed
|
|
324
|
+
|
|
325
|
+
if (effort === undefined || envelope.outputEffort) {
|
|
326
|
+
return { envelope: next, key: base, spec: baseSpec }
|
|
293
327
|
}
|
|
294
328
|
|
|
295
329
|
if (!baseSpec.thinkingEffortLevels) {
|
|
296
330
|
return {
|
|
297
|
-
|
|
331
|
+
...unresolved,
|
|
298
332
|
error: errorEvent(
|
|
299
|
-
`Model '${base}' does not support output effort (no thinkingEffortLevels). Cannot use ':${
|
|
333
|
+
`Model '${base}' does not support output effort (no thinkingEffortLevels). Cannot use ':${effort}' suffix.`,
|
|
300
334
|
'SESSION_INVALID_OUTPUT_EFFORT'
|
|
301
335
|
)
|
|
302
336
|
}
|
|
303
337
|
}
|
|
304
|
-
if (
|
|
338
|
+
if (effort !== 'none' && !baseSpec.thinkingEffortLevels[effort]) {
|
|
305
339
|
return {
|
|
306
|
-
|
|
340
|
+
...unresolved,
|
|
307
341
|
error: errorEvent(
|
|
308
|
-
`Model '${base}' does not support output effort level '${
|
|
342
|
+
`Model '${base}' does not support output effort level '${effort}'. Available: ${Object.keys(baseSpec.thinkingEffortLevels).join(', ')}`,
|
|
309
343
|
'SESSION_INVALID_OUTPUT_EFFORT'
|
|
310
344
|
)
|
|
311
345
|
}
|
|
312
346
|
}
|
|
313
347
|
|
|
314
|
-
|
|
315
|
-
|
|
348
|
+
next.outputEffort = effort
|
|
349
|
+
return { envelope: next, key: base, spec: baseSpec }
|
|
350
|
+
}
|
|
351
|
+
|
|
352
|
+
/**
|
|
353
|
+
* The two lane guards, in caller-then-internal order. Returns an
|
|
354
|
+
* error event, or `undefined` when the lane is usable.
|
|
355
|
+
*
|
|
356
|
+
* They are deliberately separate. A lane the entry does not declare is
|
|
357
|
+
* the caller asking for something this model does not sell. A lane the
|
|
358
|
+
* adapter cannot emit is the catalog and the adapter disagreeing —
|
|
359
|
+
* left to run, the call would silently take the standard lane and bill
|
|
360
|
+
* at the overlay's rates.
|
|
361
|
+
*
|
|
362
|
+
* @param {string} key
|
|
363
|
+
* @param {string} speed
|
|
364
|
+
* @param {any} spec
|
|
365
|
+
* @param {string} provider
|
|
366
|
+
* @returns {import('#core/events.js').ErrorEvent | undefined}
|
|
367
|
+
*/
|
|
368
|
+
function speedError (key, speed, spec, provider) {
|
|
369
|
+
if (!hasSpeed(spec, speed)) {
|
|
370
|
+
const available = speedNames(spec)
|
|
371
|
+
const detail = available.length
|
|
372
|
+
? `Available: ${available.join(', ')}`
|
|
373
|
+
: 'It declares no speed lanes.'
|
|
374
|
+
const hint = speed.includes(':')
|
|
375
|
+
? " Suffix order is ':effort' then '@speed'."
|
|
376
|
+
: ''
|
|
377
|
+
return errorEvent(
|
|
378
|
+
`Model '${key}' does not support speed lane '${speed}'. ${detail}${hint}`,
|
|
379
|
+
'SESSION_INVALID_SPEED'
|
|
380
|
+
)
|
|
381
|
+
}
|
|
382
|
+
if (!providerSupportsSpeed(provider)) {
|
|
383
|
+
return errorEvent(
|
|
384
|
+
`Provider '${provider}' does not implement speed lanes, but '${key}' declares '${speed}'.`,
|
|
385
|
+
'SESSION_SPEED_NOT_IMPLEMENTED'
|
|
386
|
+
)
|
|
316
387
|
}
|
|
388
|
+
return undefined
|
|
317
389
|
}
|
|
318
390
|
|
|
319
391
|
/**
|
|
@@ -332,6 +404,7 @@ function openSpan (envelope) {
|
|
|
332
404
|
}
|
|
333
405
|
if (envelope.outputBudget) attrs['gen_ai.request.max_tokens'] = envelope.outputBudget
|
|
334
406
|
if (envelope.outputEffort) attrs['mohdel.output_effort'] = envelope.outputEffort
|
|
407
|
+
if (envelope.speed) attrs['mohdel.speed'] = envelope.speed
|
|
335
408
|
return startSpan('mohdel.session.answer', attrs, parent)
|
|
336
409
|
}
|
|
337
410
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "mohdel",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.120.0",
|
|
4
4
|
"license": "MIT",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "Christophe Le Bars",
|
|
@@ -108,7 +108,7 @@
|
|
|
108
108
|
"@opentelemetry/exporter-trace-otlp-grpc": "^0.221.0",
|
|
109
109
|
"@opentelemetry/sdk-node": "^0.221.0",
|
|
110
110
|
"chalk": "^6.0.0",
|
|
111
|
-
"mohdel-thin-gate-linux-x64-gnu": "0.
|
|
111
|
+
"mohdel-thin-gate-linux-x64-gnu": "0.120.0"
|
|
112
112
|
},
|
|
113
113
|
"dependencies": {
|
|
114
114
|
"@anthropic-ai/sdk": "^0.115.0",
|
package/src/cli/check.js
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { label, err, warn, ok } from './colors.js'
|
|
2
2
|
import providers from '../lib/providers.js'
|
|
3
3
|
import { validate, isValidTag } from '../lib/schema.js'
|
|
4
|
+
import { providerSupportsSpeed } from '../../js/session/adapters/_speed.js'
|
|
4
5
|
import { getCuratedModels, loadDefaultEnv, catalogEntries, catalogValues } from '../lib/common.js'
|
|
5
6
|
|
|
6
7
|
// --- Local validation ---
|
|
@@ -50,6 +51,22 @@ const checkLocal = (curated) => {
|
|
|
50
51
|
warnings.push(`${key}: has thinkingEffortLevels but no defaultThinkingEffort`)
|
|
51
52
|
}
|
|
52
53
|
|
|
54
|
+
if (spec.speeds && !providerSupportsSpeed(keyProvider)) {
|
|
55
|
+
errors.push(`${key}: declares speeds but provider '${keyProvider}' has no adapter support — calls on those lanes would fail at dispatch`)
|
|
56
|
+
}
|
|
57
|
+
for (const [lane, overlay] of Object.entries(spec.speeds || {})) {
|
|
58
|
+
for (const priceField of ['inputPrice', 'outputPrice', 'thinkingPrice']) {
|
|
59
|
+
const val = overlay[priceField]
|
|
60
|
+
if (val != null && typeof val === 'object' && val.default == null) {
|
|
61
|
+
errors.push(`${key}: speeds.${lane}.${priceField} is tiered but missing 'default' key`)
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
const priced = ['inputPrice', 'outputPrice'].some(f => overlay[f] != null)
|
|
65
|
+
if (!priced) {
|
|
66
|
+
warnings.push(`${key}: speeds.${lane} restates no prices — the lane will bill at base rates`)
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
|
|
53
70
|
if (Array.isArray(spec.tags)) {
|
|
54
71
|
for (const t of spec.tags) {
|
|
55
72
|
if (!isValidTag(t)) warnings.push(`${key}: invalid tag "${t}" — must match /^[a-zA-Z][a-zA-Z0-9._-]{0,31}$/`)
|
package/src/lib/index.js
CHANGED
|
@@ -268,8 +268,35 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
|
|
|
268
268
|
// below. If `base` doesn't resolve, fall through — the full
|
|
269
269
|
// `modelId` (with colon) gets the normal lookup + "not
|
|
270
270
|
// found" error path.
|
|
271
|
+
// A curated id that itself contains `:` or `@` is a catalog
|
|
272
|
+
// key in its own right, so an exact hit wins over splitting.
|
|
273
|
+
// Fallback specs are excluded from that test on purpose:
|
|
274
|
+
// providers that synthesize them resolve any string, which
|
|
275
|
+
// would disable suffix parsing for them entirely.
|
|
276
|
+
const exactId = libraryMode ? modelId : expandModelAliasSync(modelId)
|
|
277
|
+
const isCuratedId = !!catalog[exactId]
|
|
278
|
+
|
|
279
|
+
// `@speed` is parsed off first so the two suffixes are read
|
|
280
|
+
// in canonical order (`base:effort@speed`).
|
|
281
|
+
let aliasSpeed
|
|
282
|
+
const atIdx = isCuratedId ? -1 : modelId.lastIndexOf('@')
|
|
283
|
+
if (atIdx > 0) {
|
|
284
|
+
const candidate = modelId.slice(atIdx + 1)
|
|
285
|
+
const base = modelId.slice(0, atIdx)
|
|
286
|
+
// The base may still carry `:effort`, which is stripped
|
|
287
|
+
// separately below — probe without it so `x:high@fast`
|
|
288
|
+
// resolves the same as `x@fast`.
|
|
289
|
+
const colon = base.lastIndexOf(':')
|
|
290
|
+
const probe = colon > 0 ? base.slice(0, colon) : base
|
|
291
|
+
const probeResolved = libraryMode ? probe : expandModelAliasSync(probe)
|
|
292
|
+
if (catalog[probeResolved] || createFallbackModelSpec(probeResolved)) {
|
|
293
|
+
aliasSpeed = candidate
|
|
294
|
+
modelId = base
|
|
295
|
+
}
|
|
296
|
+
}
|
|
297
|
+
|
|
271
298
|
let aliasOutputEffort
|
|
272
|
-
const colonIdx = modelId.lastIndexOf(':')
|
|
299
|
+
const colonIdx = isCuratedId ? -1 : modelId.lastIndexOf(':')
|
|
273
300
|
if (colonIdx > 0) {
|
|
274
301
|
const candidate = modelId.slice(colonIdx + 1)
|
|
275
302
|
const base = modelId.slice(0, colonIdx)
|
|
@@ -327,6 +354,14 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
|
|
|
327
354
|
}
|
|
328
355
|
modelSpec = normalizeModelSpec(resolvedModelId, modelSpec, providerConfig)
|
|
329
356
|
|
|
357
|
+
if (aliasSpeed && !Object.hasOwn(modelSpec.speeds || {}, aliasSpeed)) {
|
|
358
|
+
const available = Object.keys(modelSpec.speeds || {})
|
|
359
|
+
const detail = available.length
|
|
360
|
+
? `Available: ${available.join(', ')}`
|
|
361
|
+
: 'It declares no speed lanes.'
|
|
362
|
+
throw new Error(`Model '${resolvedModelId}' does not support speed lane '${aliasSpeed}'. ${detail}`)
|
|
363
|
+
}
|
|
364
|
+
|
|
330
365
|
// Validate outputEffort alias against model capabilities
|
|
331
366
|
if (aliasOutputEffort) {
|
|
332
367
|
if (!modelSpec.thinkingEffortLevels) {
|
|
@@ -337,7 +372,7 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
|
|
|
337
372
|
}
|
|
338
373
|
}
|
|
339
374
|
|
|
340
|
-
return createModelProxy(resolvedModelId, modelSpec, handlers, aliasOutputEffort, sdkCache, rateLimiter, providersConfig, cooldown, configurations, resolveProviderLimits)
|
|
375
|
+
return createModelProxy(resolvedModelId, modelSpec, handlers, aliasOutputEffort, aliasSpeed, sdkCache, rateLimiter, providersConfig, cooldown, configurations, resolveProviderLimits)
|
|
341
376
|
}
|
|
342
377
|
}
|
|
343
378
|
|
|
@@ -383,7 +418,7 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
|
|
|
383
418
|
})
|
|
384
419
|
}
|
|
385
420
|
|
|
386
|
-
const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffort, sdkCache, rateLimiter, providersConfig, cooldown, externalConfigurations, resolveProviderLimits) => {
|
|
421
|
+
const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffort, aliasSpeed, sdkCache, rateLimiter, providersConfig, cooldown, externalConfigurations, resolveProviderLimits) => {
|
|
387
422
|
// modelSpec is the full metadata object for resolvedModelId
|
|
388
423
|
let runtimePromise = null
|
|
389
424
|
|
|
@@ -445,6 +480,9 @@ const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffor
|
|
|
445
480
|
if (aliasOutputEffort && !options.outputEffort) {
|
|
446
481
|
options.outputEffort = aliasOutputEffort
|
|
447
482
|
}
|
|
483
|
+
if (aliasSpeed && !options.speed) {
|
|
484
|
+
options.speed = aliasSpeed
|
|
485
|
+
}
|
|
448
486
|
|
|
449
487
|
// Verbosity tier — captured from handlers (set by the mohdel factory). Used
|
|
450
488
|
// to gate per-call log lines. Read once per call so the closure doesn't have
|
package/src/lib/schema.js
CHANGED
|
@@ -1,3 +1,23 @@
|
|
|
1
|
+
import { SPEED_OVERRIDABLE } from '../../js/session/adapters/_speed.js'
|
|
2
|
+
|
|
3
|
+
const validateSpeeds = (speeds) => {
|
|
4
|
+
const allowed = new Set(['wire', ...SPEED_OVERRIDABLE])
|
|
5
|
+
for (const [lane, overlay] of Object.entries(speeds)) {
|
|
6
|
+
if (typeof overlay !== 'object' || overlay === null || Array.isArray(overlay)) {
|
|
7
|
+
return `lane '${lane}' must be an object`
|
|
8
|
+
}
|
|
9
|
+
if (typeof overlay.wire !== 'string' || !overlay.wire) {
|
|
10
|
+
return `lane '${lane}' must set a string 'wire' value`
|
|
11
|
+
}
|
|
12
|
+
for (const field of Object.keys(overlay)) {
|
|
13
|
+
if (!allowed.has(field)) {
|
|
14
|
+
return `lane '${lane}' may not override '${field}' (allowed: ${[...allowed].join(', ')})`
|
|
15
|
+
}
|
|
16
|
+
}
|
|
17
|
+
}
|
|
18
|
+
return null
|
|
19
|
+
}
|
|
20
|
+
|
|
1
21
|
const fieldDefs = {
|
|
2
22
|
model: { type: 'string', required: true },
|
|
3
23
|
provider: { type: 'string' },
|
|
@@ -15,6 +35,7 @@ const fieldDefs = {
|
|
|
15
35
|
thinkingTokenLimit: { type: 'number' },
|
|
16
36
|
thinkingEffortLevels: { type: 'object', nullable: true, default: null },
|
|
17
37
|
defaultThinkingEffort: { type: 'string' },
|
|
38
|
+
speeds: { type: 'object', validate: validateSpeeds },
|
|
18
39
|
tags: { type: 'array', itemType: 'string', default: [] },
|
|
19
40
|
aliases: { type: 'array', itemType: 'string', default: [] },
|
|
20
41
|
replaces: { type: 'array', itemType: 'string', default: [] },
|