mohdel 0.119.0 → 0.121.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/js/session/run.js CHANGED
@@ -26,7 +26,8 @@ import { getAdapter } from './adapters/index.js'
26
26
  import { isImageProvider } from './adapters/image/index.js'
27
27
  import { getSpec } from './adapters/_catalog.js'
28
28
  import { getProviderLimits } from './adapters/_providers.js'
29
- import { providerOf, catalogKey, effortOf } from '#core/model-id.js'
29
+ import { hasSpeed, mergeSpeed, speedHasOwnQuota, speedNames } from './adapters/_speed.js'
30
+ import { providerOf, catalogKey, effortOf, speedOf } from '#core/model-id.js'
30
31
  import * as defaultCooldown from './_cooldown.js'
31
32
  import * as defaultLimiter from './_rate_limiter.js'
32
33
  import { withIdleHeartbeat, MIN_IDLE_HEARTBEAT_MS } from './_idle_heartbeat.js'
@@ -66,15 +67,13 @@ export async function * run (envelope, {
66
67
  sleep = defaultSleep,
67
68
  signal
68
69
  } = {}) {
69
- // Honor the `model:effort` shortcut on the wire (mirrors the
70
- // factory-side `mohdel().use('model:effort')` convenience). If
71
- // the envelope's `model` field ends in `:<effort>` and the base
72
- // resolves to a known spec, split the suffix into
73
- // `envelope.outputEffort`. Explicit `outputEffort` wins when both
74
- // are set (suffix is a shortcut, not an override).
75
- const effortNorm = normalizeModelEffort(envelope, resolveSpec)
76
- if (effortNorm.error) { yield effortNorm.error; return }
77
- envelope = effortNorm.envelope
70
+ // Honor the `model:effort@speed` shortcuts on the wire (mirrors the
71
+ // factory-side `mohdel().use('model:effort')` convenience). Explicit
72
+ // envelope fields win when both are set (a suffix is a shortcut, not
73
+ // an override).
74
+ const norm = normalizeModelId(envelope, resolveSpec)
75
+ if (norm.error) { yield norm.error; return }
76
+ envelope = norm.envelope
78
77
 
79
78
  const provider = providerOf(envelope.model)
80
79
  const span = openSpan(envelope)
@@ -85,6 +84,7 @@ export async function * run (envelope, {
85
84
  provider,
86
85
  model: envelope.model,
87
86
  effort: envelope.outputEffort ?? 'default',
87
+ speed: envelope.speed ?? null,
88
88
  outputBudget: envelope.outputBudget ?? null,
89
89
  tools: envelope.tools?.length || 0,
90
90
  images: envelope.images?.length || 0
@@ -116,11 +116,8 @@ export async function * run (envelope, {
116
116
  // Catalog is authoritative: every callable model must have a
117
117
  // spec. Without one we'd silently run the provider call with
118
118
  // defaults (no rate-limits, no budget clamps, cost=0), masking
119
- // misconfiguration in the layer that pushed the catalog. Effort
120
- // suffix is stripped for the lookup — catalog entries are keyed
121
- // by the bare `<provider>/<bare>` id, not per-effort variants.
122
- const key = catalogKey(envelope.model)
123
- const spec = resolveSpec(key)
119
+ // misconfiguration in the layer that pushed the catalog.
120
+ const { key, spec } = norm
124
121
  if (!spec) {
125
122
  const detail = `Unknown model '${key}' — not in catalog`
126
123
  const err = errorEvent(detail, 'SESSION_UNKNOWN_MODEL')
@@ -130,6 +127,17 @@ export async function * run (envelope, {
130
127
  return
131
128
  }
132
129
 
130
+ if (envelope.speed) {
131
+ const speedErr = speedError(key, envelope.speed, spec, provider, adapter)
132
+ if (speedErr) {
133
+ log.warn({ provider, speed: envelope.speed }, '[mohdel:answer] unusable speed lane')
134
+ endSpanError(span, new Error(speedErr.error.message))
135
+ yield speedErr
136
+ return
137
+ }
138
+ }
139
+ const effective = mergeSpeed(spec, envelope.speed)
140
+
133
141
  const coolErr = cooldown.coolingDownError(provider)
134
142
  if (coolErr) {
135
143
  log.debug({ provider, detail: coolErr.detail }, '[mohdel:cooldown] fast-fail')
@@ -140,9 +148,12 @@ export async function * run (envelope, {
140
148
  }
141
149
 
142
150
  const providerCfg = resolveProviderLimits(provider) || {}
143
- const rpmLimit = spec?.rpmLimit ?? providerCfg.rpmLimit
144
- const tpmLimit = spec?.tpmLimit ?? providerCfg.tpmLimit
145
- const bucketKey = (spec?.rateLimitScope === 'model') ? key : provider
151
+ const rpmLimit = effective?.rpmLimit ?? providerCfg.rpmLimit
152
+ const tpmLimit = effective?.tpmLimit ?? providerCfg.tpmLimit
153
+ const baseBucket = (spec?.rateLimitScope === 'model') ? key : provider
154
+ const bucketKey = speedHasOwnQuota(spec, envelope.speed)
155
+ ? `${key}@${envelope.speed}`
156
+ : baseBucket
146
157
 
147
158
  // `0` is a killswitch ("deny all"), not "unset"; `undefined`/`null`
148
159
  // means no limit configured for that dimension. Gate on nullability
@@ -220,8 +231,12 @@ export async function * run (envelope, {
220
231
  }
221
232
  // Surface on AnswerResult so hosts that pass the whole
222
233
  // result upstream pick it up without needing a separate
223
- // wire field.
224
- if (ev.result) ev.result.maxInterFrameMs = maxInterFrameMs
234
+ // wire field. `speed` rides along because lane prices differ,
235
+ // so cost is only meaningful attributed per (model, lane).
236
+ if (ev.result) {
237
+ ev.result.maxInterFrameMs = maxInterFrameMs
238
+ if (envelope.speed) ev.result.speed = envelope.speed
239
+ }
225
240
  finalizeSpanOk(span, ev.result, sawDelta, maxInterFrameMs)
226
241
  log.debug(summarizeDone(ev.result, startedAt), '[mohdel:answer] done')
227
242
  } else if (ev.type === 'error') {
@@ -267,53 +282,114 @@ export async function * run (envelope, {
267
282
  }
268
283
 
269
284
  /**
270
- * Split an optional `:effort` suffix from `envelope.model`. If the
271
- * base resolves to a known spec, rewrites `envelope.model` and sets
272
- * `envelope.outputEffort` (unless already set). Emits a typed error
273
- * when the suffix is present and the spec rejects it.
285
+ * Split the optional `:effort` and `@speed` suffixes from
286
+ * `envelope.model`. If the base resolves to a known spec, rewrites
287
+ * `envelope.model` and sets `envelope.outputEffort` / `envelope.speed`
288
+ * (unless already set). Emits a typed error when an effort suffix is
289
+ * present and the spec rejects it; lane validity is checked in `run`
290
+ * so that an explicitly-set `envelope.speed` goes through the same
291
+ * guard as the suffix form.
292
+ *
293
+ * Also carries out the catalog lookup, so the key the spec was found
294
+ * under is the one the rest of the call uses.
274
295
  *
275
296
  * @param {import('#core/envelope.js').CallEnvelope} envelope
276
297
  * @param {(key: string) => any} resolveSpec
277
298
  * @returns {{
278
299
  * envelope: import('#core/envelope.js').CallEnvelope,
300
+ * key: string,
301
+ * spec?: any,
279
302
  * error?: import('#core/events.js').ErrorEvent
280
303
  * }}
281
304
  */
282
- function normalizeModelEffort (envelope, resolveSpec) {
283
- const candidate = effortOf(envelope.model)
284
- if (!candidate) return { envelope }
305
+ function normalizeModelId (envelope, resolveSpec) {
306
+ // A bare id that itself contains `:` or `@` is a catalog key in its
307
+ // own right; resolving the whole string first stops it being split
308
+ // into a base plus a suffix that was never meant as one.
309
+ const whole = resolveSpec(envelope.model)
310
+ if (whole) return { envelope, key: envelope.model, spec: whole }
285
311
 
312
+ const effort = effortOf(envelope.model)
313
+ const speed = speedOf(envelope.model)
286
314
  const base = catalogKey(envelope.model)
287
315
  const baseSpec = resolveSpec(base)
288
- if (!baseSpec) return { envelope } // base not known — let full string fall through to not-found
316
+ const unresolved = { envelope, key: base, spec: baseSpec }
317
+ if (effort === undefined && speed === undefined) return unresolved
318
+ if (!baseSpec) return unresolved
289
319
 
290
- // Explicit outputEffort wins; still strip the suffix so spans/logs see the canonical id.
291
- if (envelope.outputEffort) {
292
- return { envelope: { ...envelope, model: base } }
320
+ const next = { ...envelope, model: base }
321
+ if (speed !== undefined && !envelope.speed) next.speed = speed
322
+
323
+ if (effort === undefined || envelope.outputEffort) {
324
+ return { envelope: next, key: base, spec: baseSpec }
293
325
  }
294
326
 
295
327
  if (!baseSpec.thinkingEffortLevels) {
296
328
  return {
297
- envelope,
329
+ ...unresolved,
298
330
  error: errorEvent(
299
- `Model '${base}' does not support output effort (no thinkingEffortLevels). Cannot use ':${candidate}' suffix.`,
331
+ `Model '${base}' does not support output effort (no thinkingEffortLevels). Cannot use ':${effort}' suffix.`,
300
332
  'SESSION_INVALID_OUTPUT_EFFORT'
301
333
  )
302
334
  }
303
335
  }
304
- if (candidate !== 'none' && !baseSpec.thinkingEffortLevels[candidate]) {
336
+ if (effort !== 'none' && !baseSpec.thinkingEffortLevels[effort]) {
305
337
  return {
306
- envelope,
338
+ ...unresolved,
307
339
  error: errorEvent(
308
- `Model '${base}' does not support output effort level '${candidate}'. Available: ${Object.keys(baseSpec.thinkingEffortLevels).join(', ')}`,
340
+ `Model '${base}' does not support output effort level '${effort}'. Available: ${Object.keys(baseSpec.thinkingEffortLevels).join(', ')}`,
309
341
  'SESSION_INVALID_OUTPUT_EFFORT'
310
342
  )
311
343
  }
312
344
  }
313
345
 
314
- return {
315
- envelope: { ...envelope, model: base, outputEffort: candidate }
346
+ next.outputEffort = effort
347
+ return { envelope: next, key: base, spec: baseSpec }
348
+ }
349
+
350
+ /**
351
+ * The two lane guards, in caller-then-internal order. Returns an
352
+ * error event, or `undefined` when the lane is usable.
353
+ *
354
+ * They are deliberately separate. A lane the entry does not declare is
355
+ * the caller asking for something this model does not sell. A lane the
356
+ * adapter cannot emit is the catalog and the adapter disagreeing —
357
+ * left to run, the call would silently take the standard lane and bill
358
+ * at the overlay's rates.
359
+ *
360
+ * @param {string} key
361
+ * @param {string} speed
362
+ * @param {any} spec
363
+ * @param {string} provider
364
+ * @param {any} adapter
365
+ * @returns {import('#core/events.js').ErrorEvent | undefined}
366
+ */
367
+ function speedError (key, speed, spec, provider, adapter) {
368
+ if (!hasSpeed(spec, speed)) {
369
+ const available = speedNames(spec)
370
+ const detail = available.length
371
+ ? `Available: ${available.join(', ')}`
372
+ : 'It declares no speed lanes.'
373
+ const colon = speed.indexOf(':')
374
+ const hint = colon < 0
375
+ ? ''
376
+ : ` Suffix order is ':effort' then '@speed' — did you mean '${key}:${speed.slice(colon + 1)}@${speed.slice(0, colon)}'?`
377
+ return errorEvent(
378
+ `Model '${key}' does not support speed lane '${speed}'. ${detail}${hint}`,
379
+ 'SESSION_INVALID_SPEED'
380
+ )
381
+ }
382
+ const lanes = adapter?.speedLanes
383
+ if (!lanes?.has(speed)) {
384
+ const detail = lanes
385
+ ? `It accepts: ${[...lanes].join(', ')}.`
386
+ : 'It implements no speed lanes.'
387
+ return errorEvent(
388
+ `Provider '${provider}' cannot serve speed lane '${speed}', but '${key}' declares it. ${detail}`,
389
+ 'SESSION_SPEED_NOT_IMPLEMENTED'
390
+ )
316
391
  }
392
+ return undefined
317
393
  }
318
394
 
319
395
  /**
@@ -332,6 +408,7 @@ function openSpan (envelope) {
332
408
  }
333
409
  if (envelope.outputBudget) attrs['gen_ai.request.max_tokens'] = envelope.outputBudget
334
410
  if (envelope.outputEffort) attrs['mohdel.output_effort'] = envelope.outputEffort
411
+ if (envelope.speed) attrs['mohdel.speed'] = envelope.speed
335
412
  return startSpan('mohdel.session.answer', attrs, parent)
336
413
  }
337
414
 
@@ -383,6 +460,8 @@ function finalizeSpanOk (span, result, sawDelta = false, maxInterFrameMs = 0) {
383
460
  }
384
461
  if (result?.cacheWriteInputTokens) attrs['mohdel.cache_write_input_tokens'] = result.cacheWriteInputTokens
385
462
  if (result?.cacheReadInputTokens) attrs['mohdel.cache_read_input_tokens'] = result.cacheReadInputTokens
463
+ if (result?.speed) attrs['mohdel.speed'] = result.speed
464
+ if (result?.servedSpeed !== undefined) attrs['mohdel.served_speed'] = result.servedSpeed ?? 'standard'
386
465
  if (result?.cost != null) attrs['mohdel.cost'] = result.cost
387
466
  if (result?.warning) attrs['mohdel.warning'] = result.warning
388
467
  if (result?.timestamps?.start && result?.timestamps?.first) {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mohdel",
3
- "version": "0.119.0",
3
+ "version": "0.121.0",
4
4
  "license": "MIT",
5
5
  "author": {
6
6
  "name": "Christophe Le Bars",
@@ -108,12 +108,12 @@
108
108
  "@opentelemetry/exporter-trace-otlp-grpc": "^0.221.0",
109
109
  "@opentelemetry/sdk-node": "^0.221.0",
110
110
  "chalk": "^6.0.0",
111
- "mohdel-thin-gate-linux-x64-gnu": "0.119.0"
111
+ "mohdel-thin-gate-linux-x64-gnu": "0.121.0"
112
112
  },
113
113
  "dependencies": {
114
- "@anthropic-ai/sdk": "^0.115.0",
114
+ "@anthropic-ai/sdk": "^0.117.1",
115
115
  "@cerebras/cerebras_cloud_sdk": "^1.91.0",
116
- "@google/genai": "^2.16.0",
116
+ "@google/genai": "^2.17.1",
117
117
  "@opentelemetry/api": "^1.9.1",
118
118
  "env-paths": "^4.0.0",
119
119
  "groq-sdk": "^1.5.0",
@@ -126,7 +126,7 @@
126
126
  "devDependencies": {
127
127
  "gpt-tokenizer": "^3.4.0",
128
128
  "lint-staged": "^17.3.0",
129
- "release-it": "^21.0.1",
129
+ "release-it": "^21.0.2",
130
130
  "standard": "^17.1.2",
131
131
  "vitest": "^4.1.10"
132
132
  }
package/src/cli/ask.js CHANGED
@@ -176,6 +176,12 @@ Examples:
176
176
  if (tokens.outputTokens) summary.push(`${tokens.outputTokens} out`)
177
177
  if (tokens.thinkingTokens) summary.push(`${tokens.thinkingTokens} think`)
178
178
  if (tokens.cost != null) summary.push(`$${tokens.cost.toFixed(4)}`)
179
+ if (tokens.speed) {
180
+ const served = tokens.servedSpeed
181
+ summary.push(served === tokens.speed
182
+ ? `${tokens.speed} lane`
183
+ : `${tokens.speed} lane → ${served ?? 'standard'}`)
184
+ }
179
185
  const ts = tokens.timestamps
180
186
  if (ts) {
181
187
  const toMs = (a, b) => {
package/src/cli/check.js CHANGED
@@ -1,6 +1,7 @@
1
1
  import { label, err, warn, ok } from './colors.js'
2
2
  import providers from '../lib/providers.js'
3
3
  import { validate, isValidTag } from '../lib/schema.js'
4
+ import { adapters } from '../../js/session/adapters/index.js'
4
5
  import { getCuratedModels, loadDefaultEnv, catalogEntries, catalogValues } from '../lib/common.js'
5
6
 
6
7
  // --- Local validation ---
@@ -50,6 +51,24 @@ const checkLocal = (curated) => {
50
51
  warnings.push(`${key}: has thinkingEffortLevels but no defaultThinkingEffort`)
51
52
  }
52
53
 
54
+ const lanes = adapters[keyProvider]?.speedLanes
55
+ for (const [lane, overlay] of Object.entries(spec.speeds || {})) {
56
+ if (!lanes?.has(lane)) {
57
+ const detail = lanes ? `accepts: ${[...lanes].join(', ')}` : 'implements no speed lanes'
58
+ errors.push(`${key}: speeds.${lane} — provider '${keyProvider}' ${detail}; calls on that lane would fail at dispatch`)
59
+ }
60
+ for (const priceField of ['inputPrice', 'outputPrice', 'thinkingPrice']) {
61
+ const val = overlay[priceField]
62
+ if (val != null && typeof val === 'object' && val.default == null) {
63
+ errors.push(`${key}: speeds.${lane}.${priceField} is tiered but missing 'default' key`)
64
+ }
65
+ }
66
+ const priced = ['inputPrice', 'outputPrice'].some(f => overlay[f] != null)
67
+ if (!priced) {
68
+ warnings.push(`${key}: speeds.${lane} restates no prices — the lane will bill at base rates`)
69
+ }
70
+ }
71
+
53
72
  if (Array.isArray(spec.tags)) {
54
73
  for (const t of spec.tags) {
55
74
  if (!isValidTag(t)) warnings.push(`${key}: invalid tag "${t}" — must match /^[a-zA-Z][a-zA-Z0-9._-]{0,31}$/`)
package/src/lib/index.js CHANGED
@@ -75,6 +75,17 @@ const resolvePrice = (price, inputTokens) => {
75
75
  // @internal — exported for unit tests only.
76
76
  export { resolvePrice as _resolvePriceForTests }
77
77
 
78
+ // A lane candidate containing `:` means the suffixes were written the
79
+ // other way round, which is otherwise reported as an unknown lane
80
+ // spelled `fast:none`.
81
+ const reversedSuffixHint = (base, candidate) => {
82
+ const colon = candidate.indexOf(':')
83
+ if (colon < 0) return ''
84
+ const speed = candidate.slice(0, colon)
85
+ const effort = candidate.slice(colon + 1)
86
+ return ` Suffix order is ':effort' then '@speed' — did you mean '${base}:${effort}@${speed}'?`
87
+ }
88
+
78
89
  const normalizeModelSpec = (resolvedModelId, modelSpec, providerConfig) => {
79
90
  const normalized = { ...modelSpec }
80
91
  if (!normalized.provider) {
@@ -268,8 +279,35 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
268
279
  // below. If `base` doesn't resolve, fall through — the full
269
280
  // `modelId` (with colon) gets the normal lookup + "not
270
281
  // found" error path.
282
+ // A curated id that itself contains `:` or `@` is a catalog
283
+ // key in its own right, so an exact hit wins over splitting.
284
+ // Fallback specs are excluded from that test on purpose:
285
+ // providers that synthesize them resolve any string, which
286
+ // would disable suffix parsing for them entirely.
287
+ const exactId = libraryMode ? modelId : expandModelAliasSync(modelId)
288
+ const isCuratedId = !!catalog[exactId]
289
+
290
+ // `@speed` is parsed off first so the two suffixes are read
291
+ // in canonical order (`base:effort@speed`).
292
+ let aliasSpeed
293
+ const atIdx = isCuratedId ? -1 : modelId.lastIndexOf('@')
294
+ if (atIdx > 0) {
295
+ const candidate = modelId.slice(atIdx + 1)
296
+ const base = modelId.slice(0, atIdx)
297
+ // The base may still carry `:effort`, which is stripped
298
+ // separately below — probe without it so `x:high@fast`
299
+ // resolves the same as `x@fast`.
300
+ const colon = base.lastIndexOf(':')
301
+ const probe = colon > 0 ? base.slice(0, colon) : base
302
+ const probeResolved = libraryMode ? probe : expandModelAliasSync(probe)
303
+ if (catalog[probeResolved] || createFallbackModelSpec(probeResolved)) {
304
+ aliasSpeed = candidate
305
+ modelId = base
306
+ }
307
+ }
308
+
271
309
  let aliasOutputEffort
272
- const colonIdx = modelId.lastIndexOf(':')
310
+ const colonIdx = isCuratedId ? -1 : modelId.lastIndexOf(':')
273
311
  if (colonIdx > 0) {
274
312
  const candidate = modelId.slice(colonIdx + 1)
275
313
  const base = modelId.slice(0, colonIdx)
@@ -327,6 +365,17 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
327
365
  }
328
366
  modelSpec = normalizeModelSpec(resolvedModelId, modelSpec, providerConfig)
329
367
 
368
+ if (aliasSpeed && !Object.hasOwn(modelSpec.speeds || {}, aliasSpeed)) {
369
+ const available = Object.keys(modelSpec.speeds || {})
370
+ const detail = available.length
371
+ ? `Available: ${available.join(', ')}.`
372
+ : 'It declares no speed lanes.'
373
+ throw new Error(
374
+ `Model '${resolvedModelId}' does not support speed lane '${aliasSpeed}'. ${detail}` +
375
+ reversedSuffixHint(resolvedModelId, aliasSpeed)
376
+ )
377
+ }
378
+
330
379
  // Validate outputEffort alias against model capabilities
331
380
  if (aliasOutputEffort) {
332
381
  if (!modelSpec.thinkingEffortLevels) {
@@ -337,7 +386,7 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
337
386
  }
338
387
  }
339
388
 
340
- return createModelProxy(resolvedModelId, modelSpec, handlers, aliasOutputEffort, sdkCache, rateLimiter, providersConfig, cooldown, configurations, resolveProviderLimits)
389
+ return createModelProxy(resolvedModelId, modelSpec, handlers, aliasOutputEffort, aliasSpeed, sdkCache, rateLimiter, providersConfig, cooldown, configurations, resolveProviderLimits)
341
390
  }
342
391
  }
343
392
 
@@ -383,7 +432,7 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
383
432
  })
384
433
  }
385
434
 
386
- const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffort, sdkCache, rateLimiter, providersConfig, cooldown, externalConfigurations, resolveProviderLimits) => {
435
+ const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffort, aliasSpeed, sdkCache, rateLimiter, providersConfig, cooldown, externalConfigurations, resolveProviderLimits) => {
387
436
  // modelSpec is the full metadata object for resolvedModelId
388
437
  let runtimePromise = null
389
438
 
@@ -445,6 +494,9 @@ const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffor
445
494
  if (aliasOutputEffort && !options.outputEffort) {
446
495
  options.outputEffort = aliasOutputEffort
447
496
  }
497
+ if (aliasSpeed && !options.speed) {
498
+ options.speed = aliasSpeed
499
+ }
448
500
 
449
501
  // Verbosity tier — captured from handlers (set by the mohdel factory). Used
450
502
  // to gate per-call log lines. Read once per call so the closure doesn't have
package/src/lib/schema.js CHANGED
@@ -1,3 +1,20 @@
1
+ import { SPEED_OVERRIDABLE } from '../../js/session/adapters/_speed.js'
2
+
3
+ const validateSpeeds = (speeds) => {
4
+ const allowed = new Set(SPEED_OVERRIDABLE)
5
+ for (const [lane, overlay] of Object.entries(speeds)) {
6
+ if (typeof overlay !== 'object' || overlay === null || Array.isArray(overlay)) {
7
+ return `lane '${lane}' must be an object`
8
+ }
9
+ for (const field of Object.keys(overlay)) {
10
+ if (!allowed.has(field)) {
11
+ return `lane '${lane}' may not override '${field}' (allowed: ${[...allowed].join(', ')})`
12
+ }
13
+ }
14
+ }
15
+ return null
16
+ }
17
+
1
18
  const fieldDefs = {
2
19
  model: { type: 'string', required: true },
3
20
  provider: { type: 'string' },
@@ -10,11 +27,15 @@ const fieldDefs = {
10
27
  inputPrice: { type: 'number', altType: 'object' },
11
28
  outputPrice: { type: 'number', altType: 'object' },
12
29
  thinkingPrice: { type: 'number', altType: 'object' },
30
+ cacheReadPrice: { type: 'number', altType: 'object' },
31
+ cacheWritePrice: { type: 'number', altType: 'object' },
32
+ cacheWrite1hPrice: { type: 'number', altType: 'object' },
13
33
  contextTokenLimit: { type: 'number' },
14
34
  outputTokenLimit: { type: 'number' },
15
35
  thinkingTokenLimit: { type: 'number' },
16
36
  thinkingEffortLevels: { type: 'object', nullable: true, default: null },
17
37
  defaultThinkingEffort: { type: 'string' },
38
+ speeds: { type: 'object', validate: validateSpeeds },
18
39
  tags: { type: 'array', itemType: 'string', default: [] },
19
40
  aliases: { type: 'array', itemType: 'string', default: [] },
20
41
  replaces: { type: 'array', itemType: 'string', default: [] },