mohdel 0.119.0 → 0.121.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,10 +2,10 @@
2
2
  * Model-id helpers.
3
3
  *
4
4
  * A mohdel model id is a single string of shape
5
- * `"<provider>/<bare>[:<effort>]"` — same on the wire and in-process.
6
- * See PROTOCOL §3. Nothing in mohdel ever holds the id in a split
7
- * object form; when the provider or bare part is needed, these
8
- * helpers return it as a substring.
5
+ * `"<provider>/<bare>[:<effort>][@<speed>]"` — same on the wire and
6
+ * in-process. See PROTOCOL §3. Nothing in mohdel ever holds the id in
7
+ * a split object form; when a part is needed, these helpers return it
8
+ * as a substring.
9
9
  *
10
10
  * Ids are validated at ingress by the gate
11
11
  * (`rust/thin-gate/src/protocol.rs::validate_ids`); these accessors
@@ -47,12 +47,8 @@ export function bareOf (model) {
47
47
  * @returns {string}
48
48
  */
49
49
  export function catalogKey (model) {
50
- const colon = model.lastIndexOf(':')
51
- const slash = model.indexOf('/')
52
- // Only treat `:` as an effort separator when it appears after the
53
- // provider slash (otherwise a model id without `/` that happens to
54
- // contain `:` would get the wrong thing stripped).
55
- return colon > slash ? model.slice(0, colon) : model
50
+ const base = beforeSuffix(model, SPEED_SIGIL)
51
+ return beforeSuffix(base, EFFORT_SIGIL)
56
52
  }
57
53
 
58
54
  /**
@@ -62,8 +58,46 @@ export function catalogKey (model) {
62
58
  * @returns {string | undefined}
63
59
  */
64
60
  export function effortOf (model) {
65
- const colon = model.lastIndexOf(':')
61
+ return afterSuffix(beforeSuffix(model, SPEED_SIGIL), EFFORT_SIGIL)
62
+ }
63
+
64
+ /**
65
+ * Speed-lane suffix, without the `@`, or `undefined` if absent.
66
+ *
67
+ * @param {string} model
68
+ * @returns {string | undefined}
69
+ */
70
+ export function speedOf (model) {
71
+ return afterSuffix(model, SPEED_SIGIL)
72
+ }
73
+
74
+ const EFFORT_SIGIL = ':'
75
+ const SPEED_SIGIL = '@'
76
+
77
+ /**
78
+ * Index of `sigil` when it separates a suffix, or `-1`. A sigil only
79
+ * separates when it falls after the provider slash, so a bare id that
80
+ * itself contains one is left whole.
81
+ *
82
+ * @param {string} model
83
+ * @param {string} sigil
84
+ * @returns {number}
85
+ */
86
+ function suffixIndex (model, sigil) {
66
87
  const slash = model.indexOf('/')
67
- if (colon <= slash) return undefined
68
- return model.slice(colon + 1)
88
+ if (slash < 0) return -1
89
+ const at = model.lastIndexOf(sigil)
90
+ return at > slash ? at : -1
91
+ }
92
+
93
+ /** @returns {string} */
94
+ function beforeSuffix (model, sigil) {
95
+ const at = suffixIndex(model, sigil)
96
+ return at < 0 ? model : model.slice(0, at)
97
+ }
98
+
99
+ /** @returns {string | undefined} */
100
+ function afterSuffix (model, sigil) {
101
+ const at = suffixIndex(model, sigil)
102
+ return at < 0 ? undefined : model.slice(at + 1)
69
103
  }
@@ -223,6 +223,7 @@ function toEnvelope ({ modelKey, configuration, prompt, options }) {
223
223
  if (options.outputType) envelope.outputType = options.outputType
224
224
  if (options.outputStyle) envelope.outputStyle = options.outputStyle
225
225
  if (options.outputEffort) envelope.outputEffort = options.outputEffort
226
+ if (options.speed) envelope.speed = options.speed
226
227
  if (options.images?.length) envelope.images = options.images
227
228
  if (options.videos?.length) envelope.videos = options.videos
228
229
  if (options.cache !== undefined) envelope.cache = options.cache
@@ -13,7 +13,6 @@
13
13
 
14
14
  import { STATUS_INCOMPLETE, WARNING_CANCELLED } from '#core/status.js'
15
15
  import { costFor } from './_pricing.js'
16
- import { catalogKey } from '#core/model-id.js'
17
16
 
18
17
  /**
19
18
  * @param {string} start hrtime-bigint-as-string at call entry
@@ -43,7 +42,7 @@ export function cancelledDone (start, first, envelope, output, inputTokens, outp
43
42
  ...(cacheWriteInputTokens > 0 && { cacheWriteInputTokens }),
44
43
  ...(cacheReadInputTokens > 0 && { cacheReadInputTokens }),
45
44
  cost: costFor(
46
- catalogKey(envelope.model),
45
+ envelope,
47
46
  { inputTokens, outputTokens, thinkingTokens: 0, cacheWriteInputTokens, cacheReadInputTokens }
48
47
  ),
49
48
  timestamps: { start, first: first ?? end, end },
@@ -12,7 +12,10 @@
12
12
 
13
13
  import envPaths from 'env-paths'
14
14
 
15
+ import { catalogKey } from '#core/model-id.js'
16
+
15
17
  import { createLazyJsonFileCache } from './_lazy_json_cache.js'
18
+ import { mergeSpeed } from './_speed.js'
16
19
 
17
20
  // `{ suffix: null }` mirrors `src/lib/common.js::CONFIG_DIR` so the
18
21
  // session subprocess reads the same `~/.config/mohdel/curated.json`
@@ -56,3 +59,16 @@ export function setCatalog (table) {
56
59
  export function getSpec (model) {
57
60
  return cache.get(model)
58
61
  }
62
+
63
+ /**
64
+ * Effective spec for a call: the catalog entry with any speed-lane
65
+ * overlay applied. Adapters resolve spec through this rather than
66
+ * `getSpec` so lane prices and quotas reach pricing and throttling
67
+ * without each adapter merging for itself.
68
+ *
69
+ * @param {import('#core/envelope.js').CallEnvelope} envelope
70
+ * @returns {any | undefined}
71
+ */
72
+ export function specFor (envelope) {
73
+ return mergeSpeed(getSpec(catalogKey(envelope.model)), envelope.speed)
74
+ }
@@ -287,7 +287,7 @@ function finalize ({ envelope, content, toolCalls, usage, finishReason, start, f
287
287
  thinkingTokens,
288
288
  ...(cachedInputTokens > 0 && { cacheReadInputTokens: cachedInputTokens }),
289
289
  cost: costFor(
290
- catalogKey(envelope.model),
290
+ envelope,
291
291
  {
292
292
  inputTokens,
293
293
  outputTokens: visibleOutputTokens,
@@ -53,6 +53,29 @@ function scrubKey (detail, key) {
53
53
  return detail.split(key).join(mask)
54
54
  }
55
55
 
56
+ const UNWRAP_DEPTH = 2
57
+
58
+ /**
59
+ * `@google/genai` sets `ApiError.message` to `JSON.stringify({error:
60
+ * {message, code, status}})`, and Gemini's own body inside it is
61
+ * itself JSON — so the sentence sits two envelopes deep.
62
+ * @param {string} str
63
+ * @returns {string}
64
+ */
65
+ function unwrapJsonMessage (str) {
66
+ let current = str
67
+ for (let i = 0; i < UNWRAP_DEPTH; i++) {
68
+ const trimmed = current.trim()
69
+ if (!trimmed.startsWith('{')) return current
70
+ let parsed
71
+ try { parsed = JSON.parse(trimmed) } catch { return current }
72
+ const inner = parsed?.error?.message ?? parsed?.message
73
+ if (typeof inner !== 'string' || !inner) return current
74
+ current = inner
75
+ }
76
+ return current
77
+ }
78
+
56
79
  /**
57
80
  * Extract a short human-readable detail from an SDK error. Trimmed
58
81
  * to `DETAIL_CAP` chars so a verbose provider body doesn't blow up
@@ -65,7 +88,7 @@ function extractDetail (err) {
65
88
  const nested = err.error?.message || err.response?.data?.error?.message
66
89
  const raw = nested || err.message
67
90
  if (!raw) return undefined
68
- const str = typeof raw === 'string' ? raw : JSON.stringify(raw)
91
+ const str = unwrapJsonMessage(typeof raw === 'string' ? raw : JSON.stringify(raw))
69
92
  return str.length > DETAIL_CAP ? str.slice(0, DETAIL_CAP) + '…' : str
70
93
  }
71
94
 
@@ -9,7 +9,7 @@
9
9
  * @module session/adapters/_pricing
10
10
  */
11
11
 
12
- import { getSpec, setCatalog } from './_catalog.js'
12
+ import { setCatalog, specFor } from './_catalog.js'
13
13
 
14
14
  /**
15
15
  * Pure cost computation from spec + usage.
@@ -104,12 +104,15 @@ function resolveTier (price, tokens) {
104
104
  }
105
105
 
106
106
  /**
107
- * @param {string} model Fully-qualified `<provider>/<model>`.
107
+ * Cost of a call, priced against the envelope's effective spec so an
108
+ * active speed lane bills at the lane's rates.
109
+ *
110
+ * @param {import('#core/envelope.js').CallEnvelope} envelope
108
111
  * @param {{inputTokens?: number, outputTokens?: number, thinkingTokens?: number}} usage
109
112
  * @returns {number}
110
113
  */
111
- export function costFor (model, usage) {
112
- return computeCost(getSpec(model), usage)
114
+ export function costFor (envelope, usage) {
115
+ return computeCost(specFor(envelope), usage)
113
116
  }
114
117
 
115
118
  /**
@@ -0,0 +1,87 @@
1
+ /**
2
+ * Service-speed lanes — catalog side.
3
+ *
4
+ * A lane is a named service speed a model sells, declared per catalog
5
+ * entry under `speeds` with the price and rate-limit fields that
6
+ * differ from the base entry. This module owns only what is common to
7
+ * every provider: what the entry declares, and what that means for
8
+ * pricing and throttling.
9
+ *
10
+ * How a lane reaches the wire, and how the served lane is read back,
11
+ * is provider protocol and lives in the adapter. An adapter that
12
+ * handles lanes advertises the names it accepts on `speedLanes`;
13
+ * absence of that property is the declaration that it handles none,
14
+ * and is what `run.js` checks before dispatch.
15
+ *
16
+ * @module session/adapters/_speed
17
+ */
18
+
19
+ /** Spec fields a lane overlay may restate. */
20
+ export const SPEED_OVERRIDABLE = Object.freeze([
21
+ 'inputPrice',
22
+ 'outputPrice',
23
+ 'thinkingPrice',
24
+ 'cacheWritePrice',
25
+ 'cacheWrite1hPrice',
26
+ 'cacheReadPrice',
27
+ 'rpmLimit',
28
+ 'tpmLimit'
29
+ ])
30
+
31
+ /**
32
+ * Whether `spec` declares `speed`. Uses `hasOwn` rather than
33
+ * truthiness so a lane sold at base prices still counts.
34
+ *
35
+ * @param {any} spec
36
+ * @param {string} speed
37
+ * @returns {boolean}
38
+ */
39
+ export function hasSpeed (spec, speed) {
40
+ return !!spec?.speeds && Object.hasOwn(spec.speeds, speed)
41
+ }
42
+
43
+ /**
44
+ * @param {any} spec
45
+ * @returns {string[]}
46
+ */
47
+ export function speedNames (spec) {
48
+ return spec?.speeds ? Object.keys(spec.speeds) : []
49
+ }
50
+
51
+ /**
52
+ * Whether the lane declares a quota of its own, which is what earns it
53
+ * a private rate-limit bucket. Lanes that don't (OpenAI's service_tier
54
+ * shares the model's TPM/RPM pool) count against the base bucket —
55
+ * giving them their own would silently double the allowance.
56
+ *
57
+ * @param {any} spec
58
+ * @param {string} [speed]
59
+ * @returns {boolean}
60
+ */
61
+ export function speedHasOwnQuota (spec, speed) {
62
+ if (!speed || !hasSpeed(spec, speed)) return false
63
+ const overlay = spec.speeds[speed]
64
+ return overlay.rpmLimit != null || overlay.tpmLimit != null
65
+ }
66
+
67
+ /**
68
+ * Spec with `speed`'s overlay applied. Unnamed fields fall through to
69
+ * the base entry.
70
+ *
71
+ * @param {any} spec
72
+ * @param {string} [speed]
73
+ * @returns {any}
74
+ * @throws when `speed` is set and `spec` does not declare it
75
+ */
76
+ export function mergeSpeed (spec, speed) {
77
+ if (!speed) return spec
78
+ if (!hasSpeed(spec, speed)) {
79
+ throw new Error(`model does not declare speed lane '${speed}'`)
80
+ }
81
+ const overlay = spec.speeds[speed]
82
+ const merged = { ...spec }
83
+ for (const field of SPEED_OVERRIDABLE) {
84
+ if (Object.hasOwn(overlay, field)) merged[field] = overlay[field]
85
+ }
86
+ return merged
87
+ }
@@ -261,7 +261,7 @@ export async function * anthropic (envelope, deps = {}) {
261
261
  ...(cacheWrite1hTokens > 0 && { cacheWrite1hInputTokens: cacheWrite1hTokens }),
262
262
  ...(cacheReadTokens > 0 && { cacheReadInputTokens: cacheReadTokens }),
263
263
  cost: costFor(
264
- catalogKey(envelope.model),
264
+ envelope,
265
265
  {
266
266
  inputTokens,
267
267
  outputTokens: messageOutputTokens,
@@ -208,7 +208,7 @@ export async function * gemini (envelope, deps = {}) {
208
208
  thinkingTokens,
209
209
  ...(cacheReadTokens > 0 && { cacheReadInputTokens: cacheReadTokens }),
210
210
  cost: costFor(
211
- catalogKey(envelope.model),
211
+ envelope,
212
212
  { inputTokens: regularInputTokens, outputTokens, thinkingTokens, cacheReadInputTokens: cacheReadTokens }
213
213
  ),
214
214
  timestamps: { start, first: first ?? end, end }
@@ -83,6 +83,8 @@ export async function * openai (envelope, deps = {}) {
83
83
  let status = STATUS_COMPLETED
84
84
  /** @type {string | undefined} */
85
85
  let warning
86
+ /** @type {string | null | undefined} */
87
+ let servedTier
86
88
 
87
89
  // Tool-call accumulation: itemId → {call_id, name, arguments}
88
90
  /** @type {Map<string, {call_id: string, name: string, arguments: string}>} */
@@ -130,6 +132,7 @@ export async function * openai (envelope, deps = {}) {
130
132
  break
131
133
 
132
134
  case 'response.completed':
135
+ servedTier = event.response?.service_tier ?? servedTier
133
136
  if (event.response?.usage) {
134
137
  inputTokens = event.response.usage.input_tokens ?? 0
135
138
  outputTokens = event.response.usage.output_tokens ?? 0
@@ -143,6 +146,7 @@ export async function * openai (envelope, deps = {}) {
143
146
  break
144
147
 
145
148
  case 'response.incomplete':
149
+ servedTier = event.response?.service_tier ?? servedTier
146
150
  status = STATUS_INCOMPLETE
147
151
  if (event.response?.incomplete_details?.reason === 'max_output_tokens') {
148
152
  warning = WARNING_INSUFFICIENT_OUTPUT_BUDGET
@@ -189,6 +193,8 @@ export async function * openai (envelope, deps = {}) {
189
193
  // simpler with the additive shape.
190
194
  const regularInputTokens = Math.max(0, inputTokens - cachedInputTokens - cacheWriteTokens)
191
195
 
196
+ const billed = billedEnvelope(envelope, servedTier, log)
197
+
192
198
  /** @type {import('#core/events.js').DoneEvent} */
193
199
  const done = {
194
200
  type: 'done',
@@ -200,8 +206,9 @@ export async function * openai (envelope, deps = {}) {
200
206
  thinkingTokens,
201
207
  ...(cacheWriteTokens > 0 && { cacheWriteInputTokens: cacheWriteTokens }),
202
208
  ...(cachedInputTokens > 0 && { cacheReadInputTokens: cachedInputTokens }),
209
+ ...(envelope.speed && { servedSpeed: billed.served }),
203
210
  cost: costFor(
204
- catalogKey(envelope.model),
211
+ billed.envelope,
205
212
  {
206
213
  inputTokens: regularInputTokens,
207
214
  outputTokens: messageOutputTokens,
@@ -220,6 +227,66 @@ export async function * openai (envelope, deps = {}) {
220
227
  yield done
221
228
  }
222
229
 
230
+ /**
231
+ * Lane names the OpenAI adapter can put on `service_tier`. `run.js`
232
+ * reads this before dispatch; a lane outside it never reaches here.
233
+ */
234
+ openai.speedLanes = new Set(['fast', 'priority', 'flex', 'scale'])
235
+
236
+ /**
237
+ * Lane the response says was served, given the one requested.
238
+ *
239
+ * OpenAI answers `service_tier: 'priority'` to a granted request for
240
+ * either premium lane, so the echo alone cannot name which was asked
241
+ * for — the request is the other half of the answer. Any other value
242
+ * is the tier that actually ran, `null` when nothing was reported.
243
+ *
244
+ * @param {string} requested
245
+ * @param {string | null | undefined} servedTier
246
+ * @returns {string | null | undefined}
247
+ */
248
+ function servedLane (requested, servedTier) {
249
+ if (servedTier == null) return undefined
250
+ if (servedTier === 'priority') {
251
+ return (requested === 'fast' || requested === 'priority') ? requested : 'priority'
252
+ }
253
+ return openai.speedLanes.has(servedTier) ? servedTier : null
254
+ }
255
+
256
+ /**
257
+ * Envelope to price the call against. OpenAI serves Standard instead
258
+ * when a premium lane is unavailable — documented behaviour above the
259
+ * ramp rate limit — so billing the requested lane would charge premium
260
+ * rates for standard service.
261
+ *
262
+ * When a lane was requested and nothing was reported back, the request
263
+ * is the only evidence available: bill it and say so.
264
+ *
265
+ * @param {import('#core/envelope.js').CallEnvelope} envelope
266
+ * @param {string | null | undefined} servedTier
267
+ * @param {any} [log]
268
+ * @returns {{envelope: import('#core/envelope.js').CallEnvelope, served: string | null}}
269
+ */
270
+ function billedEnvelope (envelope, servedTier, log) {
271
+ if (!envelope.speed) return { envelope, served: null }
272
+
273
+ const served = servedLane(envelope.speed, servedTier)
274
+ if (served === undefined) {
275
+ log?.warn(
276
+ { model: envelope.model, speed: envelope.speed },
277
+ '[mohdel:openai] no service_tier reported; billing the requested lane'
278
+ )
279
+ return { envelope, served: envelope.speed }
280
+ }
281
+ if (served === envelope.speed) return { envelope, served }
282
+
283
+ log?.warn(
284
+ { model: envelope.model, requested: envelope.speed, servedTier, served },
285
+ '[mohdel:openai] provider served a different speed lane; billing what was served'
286
+ )
287
+ return { envelope: { ...envelope, speed: served ?? undefined }, served }
288
+ }
289
+
223
290
  /**
224
291
  * @param {import('#core/envelope.js').CallEnvelope} envelope
225
292
  * @param {Array<any>} input
@@ -287,6 +354,8 @@ function buildRequest (envelope, input, instructions) {
287
354
  }
288
355
  }
289
356
 
357
+ if (envelope.speed) request.service_tier = envelope.speed
358
+
290
359
  return request
291
360
  }
292
361