mohdel 0.119.0 → 0.121.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -0
- package/config/curated.schema.json +429 -58
- package/js/core/envelope.js +8 -0
- package/js/core/model-id.js +47 -13
- package/js/factory/bridge.js +1 -0
- package/js/session/adapters/_cancelled.js +1 -2
- package/js/session/adapters/_catalog.js +16 -0
- package/js/session/adapters/_chat_completions.js +1 -1
- package/js/session/adapters/_errors.js +24 -1
- package/js/session/adapters/_pricing.js +7 -4
- package/js/session/adapters/_speed.js +87 -0
- package/js/session/adapters/anthropic.js +1 -1
- package/js/session/adapters/gemini.js +1 -1
- package/js/session/adapters/openai.js +70 -1
- package/js/session/run.js +117 -38
- package/package.json +5 -5
- package/src/cli/ask.js +6 -0
- package/src/cli/check.js +19 -0
- package/src/lib/index.js +55 -3
- package/src/lib/schema.js +21 -0
package/js/session/run.js
CHANGED
|
@@ -26,7 +26,8 @@ import { getAdapter } from './adapters/index.js'
|
|
|
26
26
|
import { isImageProvider } from './adapters/image/index.js'
|
|
27
27
|
import { getSpec } from './adapters/_catalog.js'
|
|
28
28
|
import { getProviderLimits } from './adapters/_providers.js'
|
|
29
|
-
import {
|
|
29
|
+
import { hasSpeed, mergeSpeed, speedHasOwnQuota, speedNames } from './adapters/_speed.js'
|
|
30
|
+
import { providerOf, catalogKey, effortOf, speedOf } from '#core/model-id.js'
|
|
30
31
|
import * as defaultCooldown from './_cooldown.js'
|
|
31
32
|
import * as defaultLimiter from './_rate_limiter.js'
|
|
32
33
|
import { withIdleHeartbeat, MIN_IDLE_HEARTBEAT_MS } from './_idle_heartbeat.js'
|
|
@@ -66,15 +67,13 @@ export async function * run (envelope, {
|
|
|
66
67
|
sleep = defaultSleep,
|
|
67
68
|
signal
|
|
68
69
|
} = {}) {
|
|
69
|
-
// Honor the `model:effort`
|
|
70
|
-
// factory-side `mohdel().use('model:effort')` convenience).
|
|
71
|
-
//
|
|
72
|
-
//
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
if (effortNorm.error) { yield effortNorm.error; return }
|
|
77
|
-
envelope = effortNorm.envelope
|
|
70
|
+
// Honor the `model:effort@speed` shortcuts on the wire (mirrors the
|
|
71
|
+
// factory-side `mohdel().use('model:effort')` convenience). Explicit
|
|
72
|
+
// envelope fields win when both are set (a suffix is a shortcut, not
|
|
73
|
+
// an override).
|
|
74
|
+
const norm = normalizeModelId(envelope, resolveSpec)
|
|
75
|
+
if (norm.error) { yield norm.error; return }
|
|
76
|
+
envelope = norm.envelope
|
|
78
77
|
|
|
79
78
|
const provider = providerOf(envelope.model)
|
|
80
79
|
const span = openSpan(envelope)
|
|
@@ -85,6 +84,7 @@ export async function * run (envelope, {
|
|
|
85
84
|
provider,
|
|
86
85
|
model: envelope.model,
|
|
87
86
|
effort: envelope.outputEffort ?? 'default',
|
|
87
|
+
speed: envelope.speed ?? null,
|
|
88
88
|
outputBudget: envelope.outputBudget ?? null,
|
|
89
89
|
tools: envelope.tools?.length || 0,
|
|
90
90
|
images: envelope.images?.length || 0
|
|
@@ -116,11 +116,8 @@ export async function * run (envelope, {
|
|
|
116
116
|
// Catalog is authoritative: every callable model must have a
|
|
117
117
|
// spec. Without one we'd silently run the provider call with
|
|
118
118
|
// defaults (no rate-limits, no budget clamps, cost=0), masking
|
|
119
|
-
// misconfiguration in the layer that pushed the catalog.
|
|
120
|
-
|
|
121
|
-
// by the bare `<provider>/<bare>` id, not per-effort variants.
|
|
122
|
-
const key = catalogKey(envelope.model)
|
|
123
|
-
const spec = resolveSpec(key)
|
|
119
|
+
// misconfiguration in the layer that pushed the catalog.
|
|
120
|
+
const { key, spec } = norm
|
|
124
121
|
if (!spec) {
|
|
125
122
|
const detail = `Unknown model '${key}' — not in catalog`
|
|
126
123
|
const err = errorEvent(detail, 'SESSION_UNKNOWN_MODEL')
|
|
@@ -130,6 +127,17 @@ export async function * run (envelope, {
|
|
|
130
127
|
return
|
|
131
128
|
}
|
|
132
129
|
|
|
130
|
+
if (envelope.speed) {
|
|
131
|
+
const speedErr = speedError(key, envelope.speed, spec, provider, adapter)
|
|
132
|
+
if (speedErr) {
|
|
133
|
+
log.warn({ provider, speed: envelope.speed }, '[mohdel:answer] unusable speed lane')
|
|
134
|
+
endSpanError(span, new Error(speedErr.error.message))
|
|
135
|
+
yield speedErr
|
|
136
|
+
return
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
const effective = mergeSpeed(spec, envelope.speed)
|
|
140
|
+
|
|
133
141
|
const coolErr = cooldown.coolingDownError(provider)
|
|
134
142
|
if (coolErr) {
|
|
135
143
|
log.debug({ provider, detail: coolErr.detail }, '[mohdel:cooldown] fast-fail')
|
|
@@ -140,9 +148,12 @@ export async function * run (envelope, {
|
|
|
140
148
|
}
|
|
141
149
|
|
|
142
150
|
const providerCfg = resolveProviderLimits(provider) || {}
|
|
143
|
-
const rpmLimit =
|
|
144
|
-
const tpmLimit =
|
|
145
|
-
const
|
|
151
|
+
const rpmLimit = effective?.rpmLimit ?? providerCfg.rpmLimit
|
|
152
|
+
const tpmLimit = effective?.tpmLimit ?? providerCfg.tpmLimit
|
|
153
|
+
const baseBucket = (spec?.rateLimitScope === 'model') ? key : provider
|
|
154
|
+
const bucketKey = speedHasOwnQuota(spec, envelope.speed)
|
|
155
|
+
? `${key}@${envelope.speed}`
|
|
156
|
+
: baseBucket
|
|
146
157
|
|
|
147
158
|
// `0` is a killswitch ("deny all"), not "unset"; `undefined`/`null`
|
|
148
159
|
// means no limit configured for that dimension. Gate on nullability
|
|
@@ -220,8 +231,12 @@ export async function * run (envelope, {
|
|
|
220
231
|
}
|
|
221
232
|
// Surface on AnswerResult so hosts that pass the whole
|
|
222
233
|
// result upstream pick it up without needing a separate
|
|
223
|
-
// wire field.
|
|
224
|
-
|
|
234
|
+
// wire field. `speed` rides along because lane prices differ,
|
|
235
|
+
// so cost is only meaningful attributed per (model, lane).
|
|
236
|
+
if (ev.result) {
|
|
237
|
+
ev.result.maxInterFrameMs = maxInterFrameMs
|
|
238
|
+
if (envelope.speed) ev.result.speed = envelope.speed
|
|
239
|
+
}
|
|
225
240
|
finalizeSpanOk(span, ev.result, sawDelta, maxInterFrameMs)
|
|
226
241
|
log.debug(summarizeDone(ev.result, startedAt), '[mohdel:answer] done')
|
|
227
242
|
} else if (ev.type === 'error') {
|
|
@@ -267,53 +282,114 @@ export async function * run (envelope, {
|
|
|
267
282
|
}
|
|
268
283
|
|
|
269
284
|
/**
|
|
270
|
-
* Split
|
|
271
|
-
* base resolves to a known spec, rewrites
|
|
272
|
-
* `envelope.
|
|
273
|
-
*
|
|
285
|
+
* Split the optional `:effort` and `@speed` suffixes from
|
|
286
|
+
* `envelope.model`. If the base resolves to a known spec, rewrites
|
|
287
|
+
* `envelope.model` and sets `envelope.outputEffort` / `envelope.speed`
|
|
288
|
+
* (unless already set). Emits a typed error when an effort suffix is
|
|
289
|
+
* present and the spec rejects it; lane validity is checked in `run`
|
|
290
|
+
* so that an explicitly-set `envelope.speed` goes through the same
|
|
291
|
+
* guard as the suffix form.
|
|
292
|
+
*
|
|
293
|
+
* Also carries out the catalog lookup, so the key the spec was found
|
|
294
|
+
* under is the one the rest of the call uses.
|
|
274
295
|
*
|
|
275
296
|
* @param {import('#core/envelope.js').CallEnvelope} envelope
|
|
276
297
|
* @param {(key: string) => any} resolveSpec
|
|
277
298
|
* @returns {{
|
|
278
299
|
* envelope: import('#core/envelope.js').CallEnvelope,
|
|
300
|
+
* key: string,
|
|
301
|
+
* spec?: any,
|
|
279
302
|
* error?: import('#core/events.js').ErrorEvent
|
|
280
303
|
* }}
|
|
281
304
|
*/
|
|
282
|
-
function
|
|
283
|
-
|
|
284
|
-
|
|
305
|
+
function normalizeModelId (envelope, resolveSpec) {
|
|
306
|
+
// A bare id that itself contains `:` or `@` is a catalog key in its
|
|
307
|
+
// own right; resolving the whole string first stops it being split
|
|
308
|
+
// into a base plus a suffix that was never meant as one.
|
|
309
|
+
const whole = resolveSpec(envelope.model)
|
|
310
|
+
if (whole) return { envelope, key: envelope.model, spec: whole }
|
|
285
311
|
|
|
312
|
+
const effort = effortOf(envelope.model)
|
|
313
|
+
const speed = speedOf(envelope.model)
|
|
286
314
|
const base = catalogKey(envelope.model)
|
|
287
315
|
const baseSpec = resolveSpec(base)
|
|
288
|
-
|
|
316
|
+
const unresolved = { envelope, key: base, spec: baseSpec }
|
|
317
|
+
if (effort === undefined && speed === undefined) return unresolved
|
|
318
|
+
if (!baseSpec) return unresolved
|
|
289
319
|
|
|
290
|
-
|
|
291
|
-
if (envelope.
|
|
292
|
-
|
|
320
|
+
const next = { ...envelope, model: base }
|
|
321
|
+
if (speed !== undefined && !envelope.speed) next.speed = speed
|
|
322
|
+
|
|
323
|
+
if (effort === undefined || envelope.outputEffort) {
|
|
324
|
+
return { envelope: next, key: base, spec: baseSpec }
|
|
293
325
|
}
|
|
294
326
|
|
|
295
327
|
if (!baseSpec.thinkingEffortLevels) {
|
|
296
328
|
return {
|
|
297
|
-
|
|
329
|
+
...unresolved,
|
|
298
330
|
error: errorEvent(
|
|
299
|
-
`Model '${base}' does not support output effort (no thinkingEffortLevels). Cannot use ':${
|
|
331
|
+
`Model '${base}' does not support output effort (no thinkingEffortLevels). Cannot use ':${effort}' suffix.`,
|
|
300
332
|
'SESSION_INVALID_OUTPUT_EFFORT'
|
|
301
333
|
)
|
|
302
334
|
}
|
|
303
335
|
}
|
|
304
|
-
if (
|
|
336
|
+
if (effort !== 'none' && !baseSpec.thinkingEffortLevels[effort]) {
|
|
305
337
|
return {
|
|
306
|
-
|
|
338
|
+
...unresolved,
|
|
307
339
|
error: errorEvent(
|
|
308
|
-
`Model '${base}' does not support output effort level '${
|
|
340
|
+
`Model '${base}' does not support output effort level '${effort}'. Available: ${Object.keys(baseSpec.thinkingEffortLevels).join(', ')}`,
|
|
309
341
|
'SESSION_INVALID_OUTPUT_EFFORT'
|
|
310
342
|
)
|
|
311
343
|
}
|
|
312
344
|
}
|
|
313
345
|
|
|
314
|
-
|
|
315
|
-
|
|
346
|
+
next.outputEffort = effort
|
|
347
|
+
return { envelope: next, key: base, spec: baseSpec }
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
/**
|
|
351
|
+
* The two lane guards, in caller-then-internal order. Returns an
|
|
352
|
+
* error event, or `undefined` when the lane is usable.
|
|
353
|
+
*
|
|
354
|
+
* They are deliberately separate. A lane the entry does not declare is
|
|
355
|
+
* the caller asking for something this model does not sell. A lane the
|
|
356
|
+
* adapter cannot emit is the catalog and the adapter disagreeing —
|
|
357
|
+
* left to run, the call would silently take the standard lane and bill
|
|
358
|
+
* at the overlay's rates.
|
|
359
|
+
*
|
|
360
|
+
* @param {string} key
|
|
361
|
+
* @param {string} speed
|
|
362
|
+
* @param {any} spec
|
|
363
|
+
* @param {string} provider
|
|
364
|
+
* @param {any} adapter
|
|
365
|
+
* @returns {import('#core/events.js').ErrorEvent | undefined}
|
|
366
|
+
*/
|
|
367
|
+
function speedError (key, speed, spec, provider, adapter) {
|
|
368
|
+
if (!hasSpeed(spec, speed)) {
|
|
369
|
+
const available = speedNames(spec)
|
|
370
|
+
const detail = available.length
|
|
371
|
+
? `Available: ${available.join(', ')}`
|
|
372
|
+
: 'It declares no speed lanes.'
|
|
373
|
+
const colon = speed.indexOf(':')
|
|
374
|
+
const hint = colon < 0
|
|
375
|
+
? ''
|
|
376
|
+
: ` Suffix order is ':effort' then '@speed' — did you mean '${key}:${speed.slice(colon + 1)}@${speed.slice(0, colon)}'?`
|
|
377
|
+
return errorEvent(
|
|
378
|
+
`Model '${key}' does not support speed lane '${speed}'. ${detail}${hint}`,
|
|
379
|
+
'SESSION_INVALID_SPEED'
|
|
380
|
+
)
|
|
381
|
+
}
|
|
382
|
+
const lanes = adapter?.speedLanes
|
|
383
|
+
if (!lanes?.has(speed)) {
|
|
384
|
+
const detail = lanes
|
|
385
|
+
? `It accepts: ${[...lanes].join(', ')}.`
|
|
386
|
+
: 'It implements no speed lanes.'
|
|
387
|
+
return errorEvent(
|
|
388
|
+
`Provider '${provider}' cannot serve speed lane '${speed}', but '${key}' declares it. ${detail}`,
|
|
389
|
+
'SESSION_SPEED_NOT_IMPLEMENTED'
|
|
390
|
+
)
|
|
316
391
|
}
|
|
392
|
+
return undefined
|
|
317
393
|
}
|
|
318
394
|
|
|
319
395
|
/**
|
|
@@ -332,6 +408,7 @@ function openSpan (envelope) {
|
|
|
332
408
|
}
|
|
333
409
|
if (envelope.outputBudget) attrs['gen_ai.request.max_tokens'] = envelope.outputBudget
|
|
334
410
|
if (envelope.outputEffort) attrs['mohdel.output_effort'] = envelope.outputEffort
|
|
411
|
+
if (envelope.speed) attrs['mohdel.speed'] = envelope.speed
|
|
335
412
|
return startSpan('mohdel.session.answer', attrs, parent)
|
|
336
413
|
}
|
|
337
414
|
|
|
@@ -383,6 +460,8 @@ function finalizeSpanOk (span, result, sawDelta = false, maxInterFrameMs = 0) {
|
|
|
383
460
|
}
|
|
384
461
|
if (result?.cacheWriteInputTokens) attrs['mohdel.cache_write_input_tokens'] = result.cacheWriteInputTokens
|
|
385
462
|
if (result?.cacheReadInputTokens) attrs['mohdel.cache_read_input_tokens'] = result.cacheReadInputTokens
|
|
463
|
+
if (result?.speed) attrs['mohdel.speed'] = result.speed
|
|
464
|
+
if (result?.servedSpeed !== undefined) attrs['mohdel.served_speed'] = result.servedSpeed ?? 'standard'
|
|
386
465
|
if (result?.cost != null) attrs['mohdel.cost'] = result.cost
|
|
387
466
|
if (result?.warning) attrs['mohdel.warning'] = result.warning
|
|
388
467
|
if (result?.timestamps?.start && result?.timestamps?.first) {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "mohdel",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.121.0",
|
|
4
4
|
"license": "MIT",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "Christophe Le Bars",
|
|
@@ -108,12 +108,12 @@
|
|
|
108
108
|
"@opentelemetry/exporter-trace-otlp-grpc": "^0.221.0",
|
|
109
109
|
"@opentelemetry/sdk-node": "^0.221.0",
|
|
110
110
|
"chalk": "^6.0.0",
|
|
111
|
-
"mohdel-thin-gate-linux-x64-gnu": "0.
|
|
111
|
+
"mohdel-thin-gate-linux-x64-gnu": "0.121.0"
|
|
112
112
|
},
|
|
113
113
|
"dependencies": {
|
|
114
|
-
"@anthropic-ai/sdk": "^0.
|
|
114
|
+
"@anthropic-ai/sdk": "^0.117.1",
|
|
115
115
|
"@cerebras/cerebras_cloud_sdk": "^1.91.0",
|
|
116
|
-
"@google/genai": "^2.
|
|
116
|
+
"@google/genai": "^2.17.1",
|
|
117
117
|
"@opentelemetry/api": "^1.9.1",
|
|
118
118
|
"env-paths": "^4.0.0",
|
|
119
119
|
"groq-sdk": "^1.5.0",
|
|
@@ -126,7 +126,7 @@
|
|
|
126
126
|
"devDependencies": {
|
|
127
127
|
"gpt-tokenizer": "^3.4.0",
|
|
128
128
|
"lint-staged": "^17.3.0",
|
|
129
|
-
"release-it": "^21.0.
|
|
129
|
+
"release-it": "^21.0.2",
|
|
130
130
|
"standard": "^17.1.2",
|
|
131
131
|
"vitest": "^4.1.10"
|
|
132
132
|
}
|
package/src/cli/ask.js
CHANGED
|
@@ -176,6 +176,12 @@ Examples:
|
|
|
176
176
|
if (tokens.outputTokens) summary.push(`${tokens.outputTokens} out`)
|
|
177
177
|
if (tokens.thinkingTokens) summary.push(`${tokens.thinkingTokens} think`)
|
|
178
178
|
if (tokens.cost != null) summary.push(`$${tokens.cost.toFixed(4)}`)
|
|
179
|
+
if (tokens.speed) {
|
|
180
|
+
const served = tokens.servedSpeed
|
|
181
|
+
summary.push(served === tokens.speed
|
|
182
|
+
? `${tokens.speed} lane`
|
|
183
|
+
: `${tokens.speed} lane → ${served ?? 'standard'}`)
|
|
184
|
+
}
|
|
179
185
|
const ts = tokens.timestamps
|
|
180
186
|
if (ts) {
|
|
181
187
|
const toMs = (a, b) => {
|
package/src/cli/check.js
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { label, err, warn, ok } from './colors.js'
|
|
2
2
|
import providers from '../lib/providers.js'
|
|
3
3
|
import { validate, isValidTag } from '../lib/schema.js'
|
|
4
|
+
import { adapters } from '../../js/session/adapters/index.js'
|
|
4
5
|
import { getCuratedModels, loadDefaultEnv, catalogEntries, catalogValues } from '../lib/common.js'
|
|
5
6
|
|
|
6
7
|
// --- Local validation ---
|
|
@@ -50,6 +51,24 @@ const checkLocal = (curated) => {
|
|
|
50
51
|
warnings.push(`${key}: has thinkingEffortLevels but no defaultThinkingEffort`)
|
|
51
52
|
}
|
|
52
53
|
|
|
54
|
+
const lanes = adapters[keyProvider]?.speedLanes
|
|
55
|
+
for (const [lane, overlay] of Object.entries(spec.speeds || {})) {
|
|
56
|
+
if (!lanes?.has(lane)) {
|
|
57
|
+
const detail = lanes ? `accepts: ${[...lanes].join(', ')}` : 'implements no speed lanes'
|
|
58
|
+
errors.push(`${key}: speeds.${lane} — provider '${keyProvider}' ${detail}; calls on that lane would fail at dispatch`)
|
|
59
|
+
}
|
|
60
|
+
for (const priceField of ['inputPrice', 'outputPrice', 'thinkingPrice']) {
|
|
61
|
+
const val = overlay[priceField]
|
|
62
|
+
if (val != null && typeof val === 'object' && val.default == null) {
|
|
63
|
+
errors.push(`${key}: speeds.${lane}.${priceField} is tiered but missing 'default' key`)
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
const priced = ['inputPrice', 'outputPrice'].some(f => overlay[f] != null)
|
|
67
|
+
if (!priced) {
|
|
68
|
+
warnings.push(`${key}: speeds.${lane} restates no prices — the lane will bill at base rates`)
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
|
|
53
72
|
if (Array.isArray(spec.tags)) {
|
|
54
73
|
for (const t of spec.tags) {
|
|
55
74
|
if (!isValidTag(t)) warnings.push(`${key}: invalid tag "${t}" — must match /^[a-zA-Z][a-zA-Z0-9._-]{0,31}$/`)
|
package/src/lib/index.js
CHANGED
|
@@ -75,6 +75,17 @@ const resolvePrice = (price, inputTokens) => {
|
|
|
75
75
|
// @internal — exported for unit tests only.
|
|
76
76
|
export { resolvePrice as _resolvePriceForTests }
|
|
77
77
|
|
|
78
|
+
// A lane candidate containing `:` means the suffixes were written the
|
|
79
|
+
// other way round, which is otherwise reported as an unknown lane
|
|
80
|
+
// spelled `fast:none`.
|
|
81
|
+
const reversedSuffixHint = (base, candidate) => {
|
|
82
|
+
const colon = candidate.indexOf(':')
|
|
83
|
+
if (colon < 0) return ''
|
|
84
|
+
const speed = candidate.slice(0, colon)
|
|
85
|
+
const effort = candidate.slice(colon + 1)
|
|
86
|
+
return ` Suffix order is ':effort' then '@speed' — did you mean '${base}:${effort}@${speed}'?`
|
|
87
|
+
}
|
|
88
|
+
|
|
78
89
|
const normalizeModelSpec = (resolvedModelId, modelSpec, providerConfig) => {
|
|
79
90
|
const normalized = { ...modelSpec }
|
|
80
91
|
if (!normalized.provider) {
|
|
@@ -268,8 +279,35 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
|
|
|
268
279
|
// below. If `base` doesn't resolve, fall through — the full
|
|
269
280
|
// `modelId` (with colon) gets the normal lookup + "not
|
|
270
281
|
// found" error path.
|
|
282
|
+
// A curated id that itself contains `:` or `@` is a catalog
|
|
283
|
+
// key in its own right, so an exact hit wins over splitting.
|
|
284
|
+
// Fallback specs are excluded from that test on purpose:
|
|
285
|
+
// providers that synthesize them resolve any string, which
|
|
286
|
+
// would disable suffix parsing for them entirely.
|
|
287
|
+
const exactId = libraryMode ? modelId : expandModelAliasSync(modelId)
|
|
288
|
+
const isCuratedId = !!catalog[exactId]
|
|
289
|
+
|
|
290
|
+
// `@speed` is parsed off first so the two suffixes are read
|
|
291
|
+
// in canonical order (`base:effort@speed`).
|
|
292
|
+
let aliasSpeed
|
|
293
|
+
const atIdx = isCuratedId ? -1 : modelId.lastIndexOf('@')
|
|
294
|
+
if (atIdx > 0) {
|
|
295
|
+
const candidate = modelId.slice(atIdx + 1)
|
|
296
|
+
const base = modelId.slice(0, atIdx)
|
|
297
|
+
// The base may still carry `:effort`, which is stripped
|
|
298
|
+
// separately below — probe without it so `x:high@fast`
|
|
299
|
+
// resolves the same as `x@fast`.
|
|
300
|
+
const colon = base.lastIndexOf(':')
|
|
301
|
+
const probe = colon > 0 ? base.slice(0, colon) : base
|
|
302
|
+
const probeResolved = libraryMode ? probe : expandModelAliasSync(probe)
|
|
303
|
+
if (catalog[probeResolved] || createFallbackModelSpec(probeResolved)) {
|
|
304
|
+
aliasSpeed = candidate
|
|
305
|
+
modelId = base
|
|
306
|
+
}
|
|
307
|
+
}
|
|
308
|
+
|
|
271
309
|
let aliasOutputEffort
|
|
272
|
-
const colonIdx = modelId.lastIndexOf(':')
|
|
310
|
+
const colonIdx = isCuratedId ? -1 : modelId.lastIndexOf(':')
|
|
273
311
|
if (colonIdx > 0) {
|
|
274
312
|
const candidate = modelId.slice(colonIdx + 1)
|
|
275
313
|
const base = modelId.slice(0, colonIdx)
|
|
@@ -327,6 +365,17 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
|
|
|
327
365
|
}
|
|
328
366
|
modelSpec = normalizeModelSpec(resolvedModelId, modelSpec, providerConfig)
|
|
329
367
|
|
|
368
|
+
if (aliasSpeed && !Object.hasOwn(modelSpec.speeds || {}, aliasSpeed)) {
|
|
369
|
+
const available = Object.keys(modelSpec.speeds || {})
|
|
370
|
+
const detail = available.length
|
|
371
|
+
? `Available: ${available.join(', ')}.`
|
|
372
|
+
: 'It declares no speed lanes.'
|
|
373
|
+
throw new Error(
|
|
374
|
+
`Model '${resolvedModelId}' does not support speed lane '${aliasSpeed}'. ${detail}` +
|
|
375
|
+
reversedSuffixHint(resolvedModelId, aliasSpeed)
|
|
376
|
+
)
|
|
377
|
+
}
|
|
378
|
+
|
|
330
379
|
// Validate outputEffort alias against model capabilities
|
|
331
380
|
if (aliasOutputEffort) {
|
|
332
381
|
if (!modelSpec.thinkingEffortLevels) {
|
|
@@ -337,7 +386,7 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
|
|
|
337
386
|
}
|
|
338
387
|
}
|
|
339
388
|
|
|
340
|
-
return createModelProxy(resolvedModelId, modelSpec, handlers, aliasOutputEffort, sdkCache, rateLimiter, providersConfig, cooldown, configurations, resolveProviderLimits)
|
|
389
|
+
return createModelProxy(resolvedModelId, modelSpec, handlers, aliasOutputEffort, aliasSpeed, sdkCache, rateLimiter, providersConfig, cooldown, configurations, resolveProviderLimits)
|
|
341
390
|
}
|
|
342
391
|
}
|
|
343
392
|
|
|
@@ -383,7 +432,7 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
|
|
|
383
432
|
})
|
|
384
433
|
}
|
|
385
434
|
|
|
386
|
-
const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffort, sdkCache, rateLimiter, providersConfig, cooldown, externalConfigurations, resolveProviderLimits) => {
|
|
435
|
+
const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffort, aliasSpeed, sdkCache, rateLimiter, providersConfig, cooldown, externalConfigurations, resolveProviderLimits) => {
|
|
387
436
|
// modelSpec is the full metadata object for resolvedModelId
|
|
388
437
|
let runtimePromise = null
|
|
389
438
|
|
|
@@ -445,6 +494,9 @@ const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffor
|
|
|
445
494
|
if (aliasOutputEffort && !options.outputEffort) {
|
|
446
495
|
options.outputEffort = aliasOutputEffort
|
|
447
496
|
}
|
|
497
|
+
if (aliasSpeed && !options.speed) {
|
|
498
|
+
options.speed = aliasSpeed
|
|
499
|
+
}
|
|
448
500
|
|
|
449
501
|
// Verbosity tier — captured from handlers (set by the mohdel factory). Used
|
|
450
502
|
// to gate per-call log lines. Read once per call so the closure doesn't have
|
package/src/lib/schema.js
CHANGED
|
@@ -1,3 +1,20 @@
|
|
|
1
|
+
import { SPEED_OVERRIDABLE } from '../../js/session/adapters/_speed.js'
|
|
2
|
+
|
|
3
|
+
const validateSpeeds = (speeds) => {
|
|
4
|
+
const allowed = new Set(SPEED_OVERRIDABLE)
|
|
5
|
+
for (const [lane, overlay] of Object.entries(speeds)) {
|
|
6
|
+
if (typeof overlay !== 'object' || overlay === null || Array.isArray(overlay)) {
|
|
7
|
+
return `lane '${lane}' must be an object`
|
|
8
|
+
}
|
|
9
|
+
for (const field of Object.keys(overlay)) {
|
|
10
|
+
if (!allowed.has(field)) {
|
|
11
|
+
return `lane '${lane}' may not override '${field}' (allowed: ${[...allowed].join(', ')})`
|
|
12
|
+
}
|
|
13
|
+
}
|
|
14
|
+
}
|
|
15
|
+
return null
|
|
16
|
+
}
|
|
17
|
+
|
|
1
18
|
const fieldDefs = {
|
|
2
19
|
model: { type: 'string', required: true },
|
|
3
20
|
provider: { type: 'string' },
|
|
@@ -10,11 +27,15 @@ const fieldDefs = {
|
|
|
10
27
|
inputPrice: { type: 'number', altType: 'object' },
|
|
11
28
|
outputPrice: { type: 'number', altType: 'object' },
|
|
12
29
|
thinkingPrice: { type: 'number', altType: 'object' },
|
|
30
|
+
cacheReadPrice: { type: 'number', altType: 'object' },
|
|
31
|
+
cacheWritePrice: { type: 'number', altType: 'object' },
|
|
32
|
+
cacheWrite1hPrice: { type: 'number', altType: 'object' },
|
|
13
33
|
contextTokenLimit: { type: 'number' },
|
|
14
34
|
outputTokenLimit: { type: 'number' },
|
|
15
35
|
thinkingTokenLimit: { type: 'number' },
|
|
16
36
|
thinkingEffortLevels: { type: 'object', nullable: true, default: null },
|
|
17
37
|
defaultThinkingEffort: { type: 'string' },
|
|
38
|
+
speeds: { type: 'object', validate: validateSpeeds },
|
|
18
39
|
tags: { type: 'array', itemType: 'string', default: [] },
|
|
19
40
|
aliases: { type: 'array', itemType: 'string', default: [] },
|
|
20
41
|
replaces: { type: 'array', itemType: 'string', default: [] },
|