mohdel 1.0.3 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -422,6 +422,7 @@ OPENROUTER_API_SK=sk-or-...
422
422
  NOVITA_API_SK=...
423
423
  QWEN_API_SK=sk-...
424
424
  XIAOMI_API_SK=...
425
+ COHERE_API_SK=...
425
426
  MOHDEL_LOCAL_API_SK=...
426
427
  ```
427
428
 
@@ -452,17 +453,18 @@ What each provider supports through mohdel's unified interface:
452
453
  | Anthropic | Yes | Yes | Yes | No | Yes (adaptive / budget) | `identifier` → `metadata.user_id` |
453
454
  | OpenAI | Yes | Yes | Yes | No | Yes (o-series) | GPT-5 verbosity via `outputStyle` |
454
455
  | Gemini | Yes | Yes | Yes | Yes | Yes (`thinkingLevel` / `thinkingBudget`) | Auto-uploads large videos; content-hashed cache |
455
- | Cerebras | No | Yes | Yes | No | Yes (`reasoning_effort` or zai `disable_reasoning`) | Non-streaming chat completions |
456
- | Groq | No | Yes | Yes | No | No | Non-streaming; shared chat-completions path |
456
+ | Cerebras | Yes | Yes | Yes | No | Yes (`reasoning_effort` or zai `disable_reasoning`) | Shared chat-completions path |
457
+ | Groq | Yes | Yes | Yes | No | No | Shared chat-completions path |
457
458
  | xAI | Yes | Yes | Yes | No | Auto | OpenAI Responses API over `api.x.ai/v1` |
458
- | DeepSeek | No | Yes | Yes | No | No | DSML tool-call fallback when model emits tags in content |
459
+ | DeepSeek | No | Yes | Yes | No | No | Non-streaming: the DSML tool-call fallback is only parsed off a complete response |
459
460
  | Fireworks | Yes | Yes | Yes | No | Yes (`reasoning_effort`) | OpenAI SDK + `baseURL`; model id auto-prefixed |
460
- | Mistral | No | Yes | Yes | No | No | `tool_choice: "any"` = required |
461
- | Qwen Cloud | No | Yes | No | No | Yes (`enable_thinking` + `thinking_budget`) | Alibaba DashScope intl; hybrid models think by default — effort `none` sends explicit off |
462
- | Xiaomi | No | Yes | Yes | No | Auto | MiMo; shared chat-completions path, `reasoning_content` captured |
461
+ | Mistral | Yes | Yes | Yes | No | No | `tool_choice: "any"` = required |
462
+ | Qwen Cloud | Yes | Yes | No | No | Yes (`enable_thinking` + `thinking_budget`) | Alibaba DashScope intl; hybrid models think by default — effort `none` sends explicit off |
463
+ | Xiaomi | Yes | Yes | Yes | No | Auto | MiMo; shared chat-completions path, `reasoning_content` captured |
463
464
  | OpenRouter | Yes | Yes | Yes | No | Varies | Meta-provider; `providerOptions.openrouter` for routing prefs |
464
465
  | Local | Yes | Yes | Yes | No | No | Any OpenAI-compatible server; endpoint is the catalog entry's `baseURL` |
465
- | Novita | No | Yes | Yes | No | No | Prices in the model list; text via the shared chat-completions path, separate image adapter |
466
+ | Cohere | n/a | n/a | n/a | n/a | n/a | Embeddings only: no chat models reach mohdel through it |
467
+ | Novita | Yes | Yes | Yes | No | Yes (`reasoning_content`) | Prices in the model list; text via the shared chat-completions path, separate image adapter |
466
468
 
467
469
  Adapter capability ≠ model capability — whether a given model accepts images, tools, or thinking effort depends on the model spec in `curated.json`. The adapter passes through what the envelope supplies; the provider rejects unsupported combos.
468
470
 
@@ -517,6 +517,39 @@
517
517
  "minimum": 0,
518
518
  "description": "USD per audio minute (transcription-type entries). Token-billed transcription models (OpenAI gpt-4o-*-transcribe) use inputPrice/outputPrice instead."
519
519
  },
520
+ "embeddingPrice": {
521
+ "type": "number",
522
+ "minimum": 0,
523
+ "description": "USD per 1,000,000 input tokens for an embedding model. Without it, cost is 0 on every embed call."
524
+ },
525
+ "dimensions": {
526
+ "type": "number",
527
+ "minimum": 1,
528
+ "description": "Native width of the vectors this model returns."
529
+ },
530
+ "dimensionsSelectable": {
531
+ "type": "boolean",
532
+ "description": "Whether the model accepts a requested output width. Providers spell the parameter differently (dimensions, output_dimension, outputDimensionality) and support it on some models only; when false, passing `dimensions` fails before dispatch instead of being ignored."
533
+ },
534
+ "maxBatch": {
535
+ "type": "number",
536
+ "minimum": 1,
537
+ "description": "Inputs accepted in one embed call. 2048 on OpenAI, 96 on Cohere, 10 on Qwen Cloud, 1 for Gemini's single-content endpoint. A larger batch fails before dispatch; mohdel does not split one, because splitting changes cost attribution and result ordering."
538
+ },
539
+ "maxInputTokens": {
540
+ "type": "number",
541
+ "minimum": 1,
542
+ "description": "Token ceiling for a single input, not for the request."
543
+ },
544
+ "inputTypes": {
545
+ "type": "object",
546
+ "additionalProperties": { "type": "string" },
547
+ "description": "Maps mohdel's symbolic roles (query, document, classification, clustering) to the provider's own vocabulary, e.g. {\"query\": \"RETRIEVAL_QUERY\"} for Gemini or {\"query\": \"search_query\"} for Cohere. An entry without it rejects `inputType` rather than dropping it: the role is baked into the vector, so a silent drop produces quietly worse retrieval."
548
+ },
549
+ "defaultInputType": {
550
+ "type": "string",
551
+ "description": "Symbolic role sent when the caller names none. Required for providers that make the parameter mandatory (Cohere v3+)."
552
+ },
520
553
  "rpmLimit": {
521
554
  "type": "integer",
522
555
  "minimum": 1,
@@ -0,0 +1,50 @@
1
+ /**
2
+ * Send an EmbedEnvelope to thin-gate's `POST /v1/embed`.
3
+ *
4
+ * One-shot: a single JSON response body, no streaming, no cooldown or
5
+ * rate-limit. One call carries N inputs and returns N vectors in request
6
+ * order; the gate never splits a batch, so a batch larger than the model's
7
+ * `maxBatch` comes back as an error rather than as several calls.
8
+ *
9
+ * @module client/call_embedding
10
+ */
11
+
12
+ import { requestUnix } from './transport.js'
13
+ import { readAll, parseErrorBody } from './response.js'
14
+ import { MohdelError } from '#core'
15
+
16
+ /**
17
+ * @param {import('#core/embedding.js').EmbedEnvelope} envelope
18
+ * @param {object} options
19
+ * @param {string} options.socketPath
20
+ * @param {AbortSignal} [options.signal]
21
+ * @param {string} [options.path] HTTP path; defaults to '/v1/embed'
22
+ * @returns {Promise<import('#core/embedding.js').EmbedResult>}
23
+ */
24
+ export async function callEmbedding (envelope, { socketPath, signal, path = '/v1/embed' }) {
25
+ const res = await requestUnix({
26
+ socketPath,
27
+ path,
28
+ method: 'POST',
29
+ body: envelope,
30
+ signal
31
+ })
32
+
33
+ const body = await readAll(res)
34
+
35
+ if (res.statusCode !== 200) {
36
+ throw MohdelError.fromJSON(parseErrorBody(body, res.statusCode ?? 0))
37
+ }
38
+
39
+ let parsed
40
+ try {
41
+ parsed = JSON.parse(body)
42
+ } catch (e) {
43
+ throw new MohdelError(`gate returned an unparseable embed body: ${body.slice(0, 200)}`, {
44
+ type: 'PROTOCOL_INVALID_RESPONSE',
45
+ severity: 'error',
46
+ retryable: false
47
+ })
48
+ }
49
+ return parsed
50
+ }
@@ -16,4 +16,5 @@
16
16
  export { call } from './call.js'
17
17
  export { callImage } from './call_image.js'
18
18
  export { callTranscription } from './call_transcription.js'
19
+ export { callEmbedding } from './call_embedding.js'
19
20
  export { resolveGateBinary } from './gate-binary.js'
@@ -0,0 +1,76 @@
1
+ /**
2
+ * Embedding envelope and result.
3
+ *
4
+ * Separate call path from `CallEnvelope` / `AnswerResult`: embeddings are a
5
+ * single synchronous request/response, nothing is generated, and one call
6
+ * carries N inputs and returns N vectors in request order.
7
+ * Result shape: `{ status, vectors, dimensions, inputTokens, cost, timestamps }`.
8
+ *
9
+ * @module core/embedding
10
+ */
11
+
12
+ /**
13
+ * @typedef {object} EmbedEnvelope
14
+ *
15
+ * @property {string} callId
16
+ * @property {string} authId
17
+ * @property {import('./envelope.js').Auth} auth
18
+ * @property {string} [traceparent]
19
+ * @property {string} [baggage]
20
+ *
21
+ * @property {string} model
22
+ * Full mohdel id — `"<provider>/<bare>"`. Same shape as
23
+ * `CallEnvelope.model` (see `envelope.js`).
24
+ * @property {string[]} input
25
+ * Texts to embed. Always an array, even for one. A batch larger than the
26
+ * entry's `maxBatch` fails before dispatch rather than being split:
27
+ * splitting would change both cost attribution and result ordering.
28
+ *
29
+ * @property {number} [dimensions]
30
+ * Requested output width. Providers spell this three ways
31
+ * (`dimensions`, `output_dimension`, `outputDimensionality`) and support it
32
+ * on some models only; an entry without `dimensionsSelectable` rejects the
33
+ * field rather than sending it to be ignored.
34
+ * @property {string} [inputType]
35
+ * Symbolic role of the text: `query`, `document`, `classification`,
36
+ * `clustering`. Asymmetric models embed the same string differently
37
+ * depending on it, so a query and its matching passage land close together.
38
+ * The entry's `inputTypes` maps it to the provider's own vocabulary.
39
+ */
40
+
41
+ /**
42
+ * @typedef {object} EmbedResult
43
+ *
44
+ * @property {'completed'} status
45
+ * Embeddings are one-shot — no `incomplete` state.
46
+ * @property {number[][]} vectors
47
+ * One per input, in request order.
48
+ * @property {number} dimensions
49
+ * Width of the returned vectors, which is what the provider actually
50
+ * produced rather than what was asked for.
51
+ * @property {string | null} inputType
52
+ * The provider-native value that was sent, or null when none was. The role
53
+ * is baked into the vector, so a caller storing these must record it and
54
+ * query the same namespace the same way.
55
+ * @property {number} inputTokens
56
+ * @property {number} cost
57
+ * USD. `embeddingPrice` (per million input tokens) × `inputTokens`; 0 when
58
+ * the spec carries no price.
59
+ * @property {{start: string, first: string, end: string}} timestamps
60
+ * hrtime-bigint-as-string. `first` = `end` (no streaming).
61
+ */
62
+
63
+ export const EMBED_ENVELOPE_FIELDS = Object.freeze([
64
+ 'callId',
65
+ 'authId',
66
+ 'auth',
67
+ 'traceparent',
68
+ 'baggage',
69
+ 'model',
70
+ 'input',
71
+ 'dimensions',
72
+ 'inputType'
73
+ ])
74
+
75
+ /** Symbolic roles a caller may pass; the entry maps them to provider values. */
76
+ export const INPUT_TYPES = Object.freeze(['query', 'document', 'classification', 'clustering'])
@@ -23,6 +23,7 @@
23
23
  import { run } from '../session/run.js'
24
24
  import { runImage } from '../session/run_image.js'
25
25
  import { runTranscription } from '../session/run_transcription.js'
26
+ import { runEmbedding } from '../session/run_embedding.js'
26
27
  import { markTrustedMedia } from '../session/adapters/_media.js'
27
28
  import { MohdelError, validateIds } from '#core'
28
29
  import { createRealtimeDeltaBuffer } from '../../src/lib/utils.js'
@@ -180,6 +181,42 @@ export async function runAnswerTranscription ({ provider, model, configuration,
180
181
  return out.result
181
182
  }
182
183
 
184
+ /**
185
+ * Run an `embed()` call through the /session runtime.
186
+ *
187
+ * @param {object} args
188
+ * @param {string} args.provider
189
+ * @param {string} args.model
190
+ * @param {any} args.configuration
191
+ * @param {string | string[]} args.input One text or a batch; normalized to an
192
+ * array so the result shape never
193
+ * depends on how the caller asked.
194
+ * @param {any} [args.options] `inputType` / `dimensions` map onto
195
+ * the envelope; `callId` / `authId` are
196
+ * transport metadata.
197
+ * @param {any} [args.spec]
198
+ * @returns {Promise<any>}
199
+ */
200
+ export async function runAnswerEmbedding ({ provider, model, configuration, input, options = {}, spec }) {
201
+ const callId = options.callId || newCallId()
202
+ const authId = options.authId || 'local'
203
+ assertValidIds(callId, authId, `${provider}/${model}`)
204
+
205
+ const envelope = {
206
+ callId,
207
+ authId,
208
+ auth: configToAuth(configuration),
209
+ model: `${provider}/${model}`,
210
+ input: Array.isArray(input) ? input : [input]
211
+ }
212
+ if (options.inputType) envelope.inputType = options.inputType
213
+ if (options.dimensions !== undefined) envelope.dimensions = options.dimensions
214
+
215
+ const out = await runEmbedding(envelope, spec ? { spec } : {})
216
+ if (!out.ok) throw MohdelError.fromJSON(out.error, { provider, model })
217
+ return out.result
218
+ }
219
+
183
220
  /**
184
221
  * @param {object} args
185
222
  * @param {string} args.modelKey Mohdel catalog key `<provider>/<bare>`. The
@@ -147,6 +147,21 @@ export function computeTranscriptionCost (spec, usage) {
147
147
  return 0
148
148
  }
149
149
 
150
+ /**
151
+ * Embedding cost. `embeddingPrice` is USD per million input tokens; nothing is
152
+ * generated, so there is no output side.
153
+ *
154
+ * @param {any} spec
155
+ * @param {{inputTokens?: number}} usage
156
+ * @returns {number}
157
+ */
158
+ export function computeEmbeddingCost (spec, usage) {
159
+ if (!spec || typeof spec.embeddingPrice !== 'number') return 0
160
+ const tokens = usage.inputTokens
161
+ if (typeof tokens !== 'number' || tokens <= 0) return 0
162
+ return round((tokens / 1_000_000) * spec.embeddingPrice)
163
+ }
164
+
150
165
  /**
151
166
  * Test convenience: inject pricing-only specs by model id. Wraps
152
167
  * `setCatalog` with the `{input, output, thinking?}` shape used in
@@ -168,6 +183,9 @@ export function setPricing (table) {
168
183
  }
169
184
 
170
185
  /** @param {number} n */
186
+ // 1e-10 USD, not 1e-6. Six decimals reads as plenty until you price a short
187
+ // embedding: 4 tokens at $0.02 per million is 8e-8, which rounds to zero, and
188
+ // a million such calls then sum to zero instead of to their real cost.
171
189
  function round (n) {
172
- return Math.round(n * 1e6) / 1e6
190
+ return Math.round(n * 1e10) / 1e10
173
191
  }
@@ -17,6 +17,7 @@ import { runChatCompletions } from './_chat_completions.js'
17
17
  export async function * cerebras (envelope, deps = {}) {
18
18
  const client = deps.client ?? new Cerebras({ apiKey: envelope.auth.key })
19
19
  yield * runChatCompletions(envelope, client, {
20
+ stream: true,
20
21
  provider: 'cerebras',
21
22
  toolChoiceFlavor: 'cerebras',
22
23
  reasoningField: 'cerebras_zai'
@@ -0,0 +1,92 @@
1
+ /**
2
+ * Pieces every embedding adapter needs: the pre-dispatch checks that turn a
3
+ * provider's per-model limits into an error before a request is sent, and the
4
+ * envelope-to-provider translation of `inputType`.
5
+ *
6
+ * These fail rather than drop. A `dimensions` the model ignores, or an
7
+ * `inputType` silently discarded, both produce vectors that look fine and
8
+ * retrieve badly, which is the failure mohdel exists to prevent.
9
+ *
10
+ * @module session/adapters/embedding/shared
11
+ */
12
+
13
+ import { typedError } from '../_errors.js'
14
+
15
+ /**
16
+ * @param {import('#core/embedding.js').EmbedEnvelope} envelope
17
+ * @param {any} spec
18
+ */
19
+ export function checkBatch (envelope, spec) {
20
+ const input = envelope.input
21
+ if (!Array.isArray(input) || input.length === 0) {
22
+ throw typedError('embed requires a non-empty input array', 'EMBED_INPUT_EMPTY', false)
23
+ }
24
+ if (input.some(t => typeof t !== 'string')) {
25
+ throw typedError('embed input must be strings', 'EMBED_INPUT_INVALID', false)
26
+ }
27
+ const max = spec?.maxBatch
28
+ if (typeof max === 'number' && input.length > max) {
29
+ throw typedError(
30
+ `${input.length} inputs exceeds this model's batch limit of ${max}`,
31
+ 'EMBED_BATCH_TOO_LARGE',
32
+ false
33
+ )
34
+ }
35
+ }
36
+
37
+ /**
38
+ * @param {import('#core/embedding.js').EmbedEnvelope} envelope
39
+ * @param {any} spec
40
+ * @returns {number | undefined}
41
+ */
42
+ export function checkDimensions (envelope, spec) {
43
+ const wanted = envelope.dimensions
44
+ if (wanted === undefined) return undefined
45
+ if (!spec?.dimensionsSelectable) {
46
+ throw typedError(
47
+ 'this model returns a fixed vector width; remove `dimensions`',
48
+ 'EMBED_DIMENSIONS_UNSUPPORTED',
49
+ false
50
+ )
51
+ }
52
+ return wanted
53
+ }
54
+
55
+ /**
56
+ * Symbolic role to the provider's own vocabulary. The entry owns the mapping,
57
+ * the same way `thinkingEffortLevels` owns thinking budgets.
58
+ *
59
+ * @param {import('#core/embedding.js').EmbedEnvelope} envelope
60
+ * @param {any} spec
61
+ * @returns {string | null}
62
+ */
63
+ export function resolveInputType (envelope, spec) {
64
+ const table = spec?.inputTypes
65
+ const wanted = envelope.inputType ?? spec?.defaultInputType ?? null
66
+ if (wanted === null) return null
67
+
68
+ if (!table || typeof table !== 'object') {
69
+ throw typedError(
70
+ 'this model declares no inputTypes; remove `inputType`',
71
+ 'EMBED_INPUT_TYPE_UNSUPPORTED',
72
+ false
73
+ )
74
+ }
75
+ const native = table[wanted]
76
+ if (typeof native !== 'string') {
77
+ throw typedError(
78
+ `inputType '${wanted}' is not declared for this model; known: ${Object.keys(table).join(', ')}`,
79
+ 'EMBED_INPUT_TYPE_UNKNOWN',
80
+ false
81
+ )
82
+ }
83
+ return native
84
+ }
85
+
86
+ /**
87
+ * @param {number[][]} vectors
88
+ * @returns {number}
89
+ */
90
+ export function widthOf (vectors) {
91
+ return vectors.length && Array.isArray(vectors[0]) ? vectors[0].length : 0
92
+ }
@@ -0,0 +1,91 @@
1
+ /**
2
+ * Cohere embedding adapter.
3
+ *
4
+ * The most divergent shape mohdel talks to: `POST /v2/embed` rather than
5
+ * `/embeddings`, inputs under `texts`, `input_type` **required**, and
6
+ * `embeddings` returned as an object keyed by dtype
7
+ * (`{ float: [[...]] }`) instead of a `data[]` array. Tokens live at
8
+ * `meta.billed_units.input_tokens`.
9
+ *
10
+ * @module session/adapters/embedding/cohere
11
+ */
12
+
13
+ import { getSpec } from '../_catalog.js'
14
+ import { classifyProviderError, fromHttpStatus, typedError } from '../_errors.js'
15
+ import { computeEmbeddingCost } from '../_pricing.js'
16
+ import { catalogKey, bareOf } from '#core/model-id.js'
17
+ import { checkBatch, checkDimensions, resolveInputType, widthOf } from './_shared.js'
18
+
19
+ const BASE_URL = 'https://api.cohere.com/v2'
20
+
21
+ export async function cohereEmbedding (envelope, deps = {}) {
22
+ const fetchFn = deps.fetch ?? globalThis.fetch
23
+ const spec = deps.spec ?? getSpec(catalogKey(envelope.model)) ?? {}
24
+ const start = String(process.hrtime.bigint())
25
+
26
+ checkBatch(envelope, spec)
27
+ const dimensions = checkDimensions(envelope, spec)
28
+ const inputType = resolveInputType(envelope, spec)
29
+
30
+ // Cohere rejects a call without it, so an entry that declares no mapping is
31
+ // unusable rather than merely less accurate.
32
+ if (!inputType) {
33
+ throw typedError(
34
+ 'cohere requires an input_type; set inputType or the entry\'s defaultInputType',
35
+ 'EMBED_INPUT_TYPE_REQUIRED',
36
+ false
37
+ )
38
+ }
39
+
40
+ const body = {
41
+ model: spec.model ?? bareOf(envelope.model),
42
+ texts: envelope.input,
43
+ input_type: inputType,
44
+ embedding_types: ['float'],
45
+ ...(dimensions !== undefined ? { output_dimension: dimensions } : {})
46
+ }
47
+
48
+ let res
49
+ try {
50
+ res = await fetchFn(`${BASE_URL}/embed`, {
51
+ method: 'POST',
52
+ headers: {
53
+ 'Content-Type': 'application/json',
54
+ Authorization: `Bearer ${envelope.auth.key}`
55
+ },
56
+ body: JSON.stringify(body)
57
+ })
58
+ } catch (e) {
59
+ throw typedError(classifyProviderError(e, envelope.auth?.key).message, 'NET_ERROR', true)
60
+ }
61
+
62
+ if (!res.ok) {
63
+ const detail = await res.text().catch(() => '')
64
+ throw fromHttpStatus(res.status, detail, envelope.auth?.key)
65
+ }
66
+
67
+ const payload = await res.json()
68
+ const vectors = payload?.embeddings?.float ?? []
69
+
70
+ if (vectors.length !== envelope.input.length || vectors.some(v => !Array.isArray(v))) {
71
+ throw typedError(
72
+ `expected ${envelope.input.length} vectors, got ${vectors.length}`,
73
+ 'EMBED_RESULT_MISMATCH',
74
+ false
75
+ )
76
+ }
77
+
78
+ const inputTokens = payload?.meta?.billed_units?.input_tokens ??
79
+ payload?.meta?.tokens?.input_tokens ?? 0
80
+ const end = String(process.hrtime.bigint())
81
+
82
+ return {
83
+ status: 'completed',
84
+ vectors,
85
+ dimensions: widthOf(vectors),
86
+ inputType,
87
+ inputTokens,
88
+ cost: computeEmbeddingCost(spec, { inputTokens }),
89
+ timestamps: { start, first: end, end }
90
+ }
91
+ }
@@ -0,0 +1,92 @@
1
+ /**
2
+ * Gemini embedding adapter.
3
+ *
4
+ * Shares nothing with the OpenAI shape: a different endpoint per batch size
5
+ * (`:embedContent` for one input, `:batchEmbedContents` for several), vectors
6
+ * under `embedding.values`, and tokens under
7
+ * `usageMetadata.promptTokenCount`. The dimension parameter is
8
+ * `outputDimensionality` and lives inside a config object rather than at the
9
+ * top level.
10
+ *
11
+ * @module session/adapters/embedding/gemini
12
+ */
13
+
14
+ import { getSpec } from '../_catalog.js'
15
+ import { classifyProviderError, fromHttpStatus, typedError } from '../_errors.js'
16
+ import { computeEmbeddingCost } from '../_pricing.js'
17
+ import { catalogKey, bareOf } from '#core/model-id.js'
18
+ import { checkBatch, checkDimensions, resolveInputType, widthOf } from './_shared.js'
19
+
20
+ const BASE_URL = 'https://generativelanguage.googleapis.com/v1beta'
21
+
22
+ export async function geminiEmbedding (envelope, deps = {}) {
23
+ const fetchFn = deps.fetch ?? globalThis.fetch
24
+ const spec = deps.spec ?? getSpec(catalogKey(envelope.model)) ?? {}
25
+ const start = String(process.hrtime.bigint())
26
+
27
+ checkBatch(envelope, spec)
28
+ const dimensions = checkDimensions(envelope, spec)
29
+ const taskType = resolveInputType(envelope, spec)
30
+
31
+ const model = spec.model ?? bareOf(envelope.model)
32
+ const name = model.startsWith('models/') ? model : `models/${model}`
33
+ const batched = envelope.input.length > 1
34
+
35
+ const one = (text) => ({
36
+ model: name,
37
+ content: { parts: [{ text }] },
38
+ ...(taskType ? { taskType } : {}),
39
+ ...(dimensions !== undefined ? { outputDimensionality: dimensions } : {})
40
+ })
41
+ const body = batched
42
+ ? { requests: envelope.input.map(one) }
43
+ : one(envelope.input[0])
44
+
45
+ const method = batched ? 'batchEmbedContents' : 'embedContent'
46
+ let res
47
+ try {
48
+ res = await fetchFn(`${BASE_URL}/${name}:${method}`, {
49
+ method: 'POST',
50
+ headers: {
51
+ 'Content-Type': 'application/json',
52
+ 'x-goog-api-key': envelope.auth.key
53
+ },
54
+ body: JSON.stringify(body)
55
+ })
56
+ } catch (e) {
57
+ throw typedError(classifyProviderError(e, envelope.auth?.key).message, 'NET_ERROR', true)
58
+ }
59
+
60
+ if (!res.ok) {
61
+ const detail = await res.text().catch(() => '')
62
+ throw fromHttpStatus(res.status, detail, envelope.auth?.key)
63
+ }
64
+
65
+ const payload = await res.json()
66
+ const vectors = batched
67
+ ? (payload?.embeddings ?? []).map(e => e?.values)
68
+ : [payload?.embedding?.values]
69
+
70
+ if (vectors.length !== envelope.input.length || vectors.some(v => !Array.isArray(v))) {
71
+ throw typedError(
72
+ `expected ${envelope.input.length} vectors, got ${vectors.length}`,
73
+ 'EMBED_RESULT_MISMATCH',
74
+ false
75
+ )
76
+ }
77
+
78
+ // Only the single-content endpoint reports usage; the batch one does not,
79
+ // so a batched call prices at 0 rather than on a guessed token count.
80
+ const inputTokens = payload?.usageMetadata?.promptTokenCount ?? 0
81
+ const end = String(process.hrtime.bigint())
82
+
83
+ return {
84
+ status: 'completed',
85
+ vectors,
86
+ dimensions: widthOf(vectors),
87
+ inputType: taskType,
88
+ inputTokens,
89
+ cost: computeEmbeddingCost(spec, { inputTokens }),
90
+ timestamps: { start, first: end, end }
91
+ }
92
+ }
@@ -0,0 +1,37 @@
1
+ /**
2
+ * Embedding-adapter registry. Mirrors `session/adapters/transcription` but
3
+ * scoped to providers with an embeddings endpoint.
4
+ *
5
+ * Five of mohdel's thirteen providers offer embeddings at all, and neither
6
+ * meta-provider does. Of those that do, only the base URL and the name of the
7
+ * dimension parameter differ across the OpenAI-shaped ones, so they share one
8
+ * adapter; Gemini and Cohere need their own.
9
+ *
10
+ * @module session/adapters/embedding
11
+ */
12
+
13
+ import { createEmbeddingAdapter } from './openai_compatible.js'
14
+ import { geminiEmbedding } from './gemini.js'
15
+ import { cohereEmbedding } from './cohere.js'
16
+
17
+ const EMBEDDING_ADAPTERS = {
18
+ openai: createEmbeddingAdapter({ baseURL: 'https://api.openai.com/v1' }),
19
+ // The endpoint is the catalog entry's `baseURL`, as it is for local chat.
20
+ local: createEmbeddingAdapter(),
21
+ gemini: geminiEmbedding,
22
+ cohere: cohereEmbedding
23
+ }
24
+
25
+ /** Providers with an embeddings adapter, for capability checks. */
26
+ export const EMBEDDING_PROVIDERS = Object.freeze(Object.keys(EMBEDDING_ADAPTERS))
27
+
28
+ /**
29
+ * @param {string} provider
30
+ */
31
+ export function getEmbeddingAdapter (provider) {
32
+ const adapter = EMBEDDING_ADAPTERS[provider]
33
+ if (!adapter) throw new Error(`no embedding adapter for provider: ${provider}`)
34
+ return adapter
35
+ }
36
+
37
+ export const embeddingAdapters = Object.freeze(EMBEDDING_ADAPTERS)
@@ -0,0 +1,92 @@
1
+ /**
2
+ * Shared embedding adapter for OpenAI-compatible `POST <baseURL>/embeddings`.
3
+ *
4
+ * Covers OpenAI and any self-hosted server that implements the endpoint
5
+ * (Ollama, vLLM, llama.cpp), and is the same shape Mistral, Fireworks and
6
+ * Qwen Cloud expose when they are added. Only the base URL and the name of the
7
+ * dimension parameter differ, so both are bound per provider in `./index.js`.
8
+ *
9
+ * @module session/adapters/embedding/openai_compatible
10
+ */
11
+
12
+ import { getSpec } from '../_catalog.js'
13
+ import { classifyProviderError, fromHttpStatus, typedError } from '../_errors.js'
14
+ import { computeEmbeddingCost } from '../_pricing.js'
15
+ import { catalogKey, bareOf } from '#core/model-id.js'
16
+ import { checkBatch, checkDimensions, resolveInputType, widthOf } from './_shared.js'
17
+
18
+ /**
19
+ * @param {{baseURL?: string, dimensionsField?: string}} config
20
+ */
21
+ export function createEmbeddingAdapter ({ baseURL, dimensionsField = 'dimensions' } = {}) {
22
+ return async function embedding (envelope, deps = {}) {
23
+ const fetchFn = deps.fetch ?? globalThis.fetch
24
+ const spec = deps.spec ?? getSpec(catalogKey(envelope.model)) ?? {}
25
+ const start = String(process.hrtime.bigint())
26
+
27
+ checkBatch(envelope, spec)
28
+ const dimensions = checkDimensions(envelope, spec)
29
+ const inputType = resolveInputType(envelope, spec)
30
+
31
+ // `local/` carries its endpoint on the entry; everything else is bound
32
+ // to a base URL by the registry.
33
+ const root = (spec.baseURL ?? baseURL ?? '').replace(/\/$/, '')
34
+ if (!root) {
35
+ throw typedError('no baseURL for this embedding model', 'CONFIGURATION_MISSING', false)
36
+ }
37
+
38
+ /** @type {Record<string, any>} */
39
+ const body = { model: spec.model ?? bareOf(envelope.model), input: envelope.input }
40
+ if (dimensions !== undefined) body[dimensionsField] = dimensions
41
+ if (inputType) body.input_type = inputType
42
+
43
+ let res
44
+ try {
45
+ res = await fetchFn(`${root}/embeddings`, {
46
+ method: 'POST',
47
+ headers: {
48
+ 'Content-Type': 'application/json',
49
+ ...(envelope.auth?.key ? { Authorization: `Bearer ${envelope.auth.key}` } : {})
50
+ },
51
+ body: JSON.stringify(body)
52
+ })
53
+ } catch (e) {
54
+ throw typedError(classifyProviderError(e, envelope.auth?.key).message, 'NET_ERROR', true)
55
+ }
56
+
57
+ if (!res.ok) {
58
+ const detail = await res.text().catch(() => '')
59
+ throw fromHttpStatus(res.status, detail, envelope.auth?.key)
60
+ }
61
+
62
+ const payload = await res.json()
63
+ const rows = Array.isArray(payload?.data) ? payload.data : []
64
+ // `index` is authoritative: the caller matches vectors to inputs by
65
+ // position, and a provider is free to answer out of order.
66
+ const vectors = rows
67
+ .slice()
68
+ .sort((a, b) => (a?.index ?? 0) - (b?.index ?? 0))
69
+ .map(r => r?.embedding)
70
+
71
+ if (vectors.length !== envelope.input.length || vectors.some(v => !Array.isArray(v))) {
72
+ throw typedError(
73
+ `expected ${envelope.input.length} vectors, got ${vectors.length}`,
74
+ 'EMBED_RESULT_MISMATCH',
75
+ false
76
+ )
77
+ }
78
+
79
+ const inputTokens = payload?.usage?.prompt_tokens ?? payload?.usage?.total_tokens ?? 0
80
+ const end = String(process.hrtime.bigint())
81
+
82
+ return {
83
+ status: 'completed',
84
+ vectors,
85
+ dimensions: widthOf(vectors),
86
+ inputType,
87
+ inputTokens,
88
+ cost: computeEmbeddingCost(spec, { inputTokens }),
89
+ timestamps: { start, first: end, end }
90
+ }
91
+ }
92
+ }
@@ -1,5 +1,5 @@
1
1
  /**
2
- * Groq adapter — OpenAI-compatible chat completions, non-streaming.
2
+ * Groq adapter — OpenAI-compatible chat completions.
3
3
  *
4
4
  * @module session/adapters/groq
5
5
  */
@@ -19,7 +19,10 @@ export async function * groq (envelope, deps = {}) {
19
19
  apiKey: envelope.auth.key,
20
20
  fetchOptions: { dispatcher: streamingDispatcher() }
21
21
  })
22
- yield * runChatCompletions(envelope, client, { provider: 'groq' }, {
22
+ yield * runChatCompletions(envelope, client, {
23
+ provider: 'groq',
24
+ stream: true
25
+ }, {
23
26
  signal: deps.signal,
24
27
  log: deps.log,
25
28
  span: deps.span
@@ -26,6 +26,7 @@ export async function * mistral (envelope, deps = {}) {
26
26
  fetchOptions: { dispatcher: streamingDispatcher() }
27
27
  })
28
28
  yield * runChatCompletions(envelope, client, {
29
+ stream: true,
29
30
  provider: 'mistral',
30
31
  toolChoiceFlavor: 'mistral'
31
32
  }, {
@@ -9,6 +9,7 @@
9
9
  import OpenAI from 'openai'
10
10
 
11
11
  import { runChatCompletions } from './_chat_completions.js'
12
+ import { streamingDispatcher } from './_dispatcher.js'
12
13
 
13
14
  const BASE_URL = 'https://api.novita.ai/openai'
14
15
 
@@ -18,9 +19,14 @@ const BASE_URL = 'https://api.novita.ai/openai'
18
19
  * @returns {AsyncGenerator<import('#core/events.js').Event>}
19
20
  */
20
21
  export async function * novita (envelope, deps = {}) {
21
- const client = deps.client ?? new OpenAI({ apiKey: envelope.auth.key, baseURL: BASE_URL })
22
+ const client = deps.client ?? new OpenAI({
23
+ apiKey: envelope.auth.key,
24
+ baseURL: BASE_URL,
25
+ fetchOptions: { dispatcher: streamingDispatcher() }
26
+ })
22
27
  yield * runChatCompletions(envelope, client, {
23
- provider: 'novita'
28
+ provider: 'novita',
29
+ stream: true
24
30
  }, {
25
31
  signal: deps.signal,
26
32
  log: deps.log,
@@ -27,6 +27,7 @@ export async function * qwen (envelope, deps = {}) {
27
27
  fetchOptions: { dispatcher: streamingDispatcher() }
28
28
  })
29
29
  yield * runChatCompletions(envelope, client, {
30
+ stream: true,
30
31
  provider: 'qwen',
31
32
  reasoningField: 'qwen'
32
33
  }, {
@@ -26,6 +26,7 @@ export async function * xiaomi (envelope, deps = {}) {
26
26
  fetchOptions: { dispatcher: streamingDispatcher() }
27
27
  })
28
28
  yield * runChatCompletions(envelope, client, {
29
+ stream: true,
29
30
  provider: 'xiaomi'
30
31
  }, {
31
32
  signal: deps.signal,
@@ -18,6 +18,7 @@ import { MAX_LINE_BYTES, exceedsLineBytes } from '#core/framing.js'
18
18
  import { run } from './run.js'
19
19
  import { runImage } from './run_image.js'
20
20
  import { runTranscription } from './run_transcription.js'
21
+ import { runEmbedding } from './run_embedding.js'
21
22
  import { setCatalog } from './adapters/_catalog.js'
22
23
 
23
24
  // Bounded memory for pre-dequeue cancels. Hostile/buggy supervisors
@@ -221,6 +222,16 @@ export async function drive (stdin, stdout) {
221
222
  } else {
222
223
  await writeLine({ type: 'error', error: out.error })
223
224
  }
225
+ } else if (envelope.op === 'embed') {
226
+ // Same one-shot contract; shape matches `js/core/embedding.js`
227
+ // after the tag strip. One line carries all N vectors.
228
+ const { op: _op, ...embEnv } = envelope
229
+ const out = await runEmbedding(embEnv)
230
+ if (out.ok) {
231
+ await writeLine({ type: 'embed_done', result: out.result })
232
+ } else {
233
+ await writeLine({ type: 'error', error: out.error })
234
+ }
224
235
  } else {
225
236
  for await (const ev of run(envelope, { signal: controller.signal })) {
226
237
  await writeLine(ev)
@@ -0,0 +1,51 @@
1
+ /**
2
+ * Embedding runtime. Resolves the adapter for the envelope's provider and
3
+ * returns either a result or a typed error, never throwing.
4
+ *
5
+ * Mirrors `run_transcription.js`: one synchronous request, no streaming, no
6
+ * cancellation path beyond the caller's own signal.
7
+ *
8
+ * @module session/run_embedding
9
+ */
10
+
11
+ import { getEmbeddingAdapter } from './adapters/embedding/index.js'
12
+ import { classifyProviderError } from './adapters/_errors.js'
13
+ import { providerOf } from '#core/model-id.js'
14
+
15
+ /**
16
+ * @param {import('#core/embedding.js').EmbedEnvelope} envelope
17
+ * @param {{resolveAdapter?: (provider: string) => any, spec?: any}} [options]
18
+ * @returns {Promise<
19
+ * | {ok: true, result: import('#core/embedding.js').EmbedResult}
20
+ * | {ok: false, error: import('#core/errors.js').TypedError}
21
+ * >}
22
+ */
23
+ export async function runEmbedding (envelope, { resolveAdapter = getEmbeddingAdapter, spec } = {}) {
24
+ let adapter
25
+ try {
26
+ adapter = resolveAdapter(providerOf(envelope.model))
27
+ } catch (e) {
28
+ return {
29
+ ok: false,
30
+ error: {
31
+ message: messageOf(e),
32
+ severity: 'error',
33
+ retryable: false,
34
+ type: 'SESSION_UNKNOWN_PROVIDER'
35
+ }
36
+ }
37
+ }
38
+
39
+ try {
40
+ const result = await adapter(envelope, spec ? { spec } : {})
41
+ return { ok: true, result }
42
+ } catch (e) {
43
+ const typed = /** @type {any} */(e).typed || classifyProviderError(e, envelope.auth?.key)
44
+ return { ok: false, error: typed }
45
+ }
46
+ }
47
+
48
+ /** @param {unknown} e */
49
+ function messageOf (e) {
50
+ return e instanceof Error ? e.message : String(e)
51
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mohdel",
3
- "version": "1.0.3",
3
+ "version": "1.2.0",
4
4
  "license": "MIT",
5
5
  "author": {
6
6
  "name": "Christophe Le Bars",
@@ -135,7 +135,7 @@
135
135
  "@opentelemetry/exporter-trace-otlp-grpc": "^0.222.0",
136
136
  "@opentelemetry/sdk-node": "^0.222.0",
137
137
  "chalk": "^6.0.0",
138
- "mohdel-thin-gate-linux-x64-gnu": "1.0.3"
138
+ "mohdel-thin-gate-linux-x64-gnu": "1.2.0"
139
139
  },
140
140
  "dependencies": {
141
141
  "@anthropic-ai/sdk": "^0.125.0",
@@ -59,6 +59,18 @@ const creators = {
59
59
  logo: 'moonshotai.svg',
60
60
  description: 'Moonshot AI ships fluent, Chinese-first assistants and lean models tuned for consumer chat and business workflows.'
61
61
  },
62
+ cohere: {
63
+ prefixes: ['embed', 'command', 'rerank'],
64
+ label: 'Cohere',
65
+ logo: 'cohere.svg',
66
+ description: 'Cohere builds retrieval-focused models: embeddings and rerankers aimed at enterprise search rather than chat.'
67
+ },
68
+ nomic: {
69
+ prefixes: ['nomic-embed'],
70
+ label: 'Nomic',
71
+ logo: 'nomic.svg',
72
+ description: 'Nomic publishes open-weight embedding models with Matryoshka dimensions, widely self-hosted through Ollama and vLLM.'
73
+ },
62
74
  openai: {
63
75
  prefixes: ['gpt', 'whisper', 'dall-e', 'sora', 'text-embedding', 'o1', 'o3', 'o4'],
64
76
  label: 'OpenAI',
package/src/lib/index.js CHANGED
@@ -16,7 +16,7 @@ import {
16
16
  import { createRateLimiter } from '../../js/session/_rate_limiter.js'
17
17
  import { createCooldownTracker } from '../../js/session/_cooldown.js'
18
18
  import { setCatalog } from '../../js/session/adapters/_catalog.js'
19
- import { runAnswer, runAnswerImage, runAnswerTranscription } from '../../js/factory/bridge.js'
19
+ import { runAnswer, runAnswerEmbedding, runAnswerImage, runAnswerTranscription } from '../../js/factory/bridge.js'
20
20
  import { startSpan, endSpanOk, endSpanError } from './tracing.js'
21
21
  import { isValidTag } from './schema.js'
22
22
  import { silent } from './logger.js'
@@ -729,6 +729,20 @@ const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffor
729
729
  }
730
730
  }
731
731
 
732
+ if (prop === 'embed') {
733
+ return async (input, options = {}) => {
734
+ const { configuration } = await getRuntime()
735
+ return runAnswerEmbedding({
736
+ provider: modelSpec.provider,
737
+ model: modelSpec.model ?? resolvedModelId.split('/').pop(),
738
+ configuration,
739
+ input,
740
+ options,
741
+ spec: modelSpec
742
+ })
743
+ }
744
+ }
745
+
732
746
  if (prop === 'setRateLimit') {
733
747
  return async ({ rpm, tpm } = {}) => {
734
748
  const curatedCache = getCuratedCacheSnapshot()
@@ -86,6 +86,13 @@ const PROVIDER_INFO = {
86
86
  hint: 'Create an API key in the MiMo open platform console',
87
87
  free: false
88
88
  },
89
+ cohere: {
90
+ label: 'Cohere',
91
+ description: 'Embeddings and reranking for retrieval. No chat models through mohdel.',
92
+ url: 'https://dashboard.cohere.com/api-keys',
93
+ hint: 'Create an API key in the Cohere dashboard under API Keys',
94
+ free: true
95
+ },
89
96
  qwen: {
90
97
  label: 'Qwen Cloud',
91
98
  description: 'Qwen — reasoning, coding, long context. Free quota for new users.',
@@ -86,6 +86,18 @@ const providers = {
86
86
  contextSemantics: 'shared',
87
87
  outputCapStrategy: 'accept'
88
88
  },
89
+ cohere: {
90
+ sdk: 'cohere',
91
+ api: 'embeddings',
92
+ apiKeyEnv: 'COHERE_API_SK',
93
+ baseURL: 'https://api.cohere.com/v2',
94
+ createConfiguration: apiKey => ({ apiKey }),
95
+ references: {
96
+ pricing: 'https://cohere.com/pricing',
97
+ models: 'https://docs.cohere.com/docs/models',
98
+ rateLimits: 'https://docs.cohere.com/docs/rate-limits'
99
+ }
100
+ },
89
101
  mistral: {
90
102
  sdk: 'openai',
91
103
  api: 'chatCompletions',
package/src/lib/schema.js CHANGED
@@ -58,6 +58,13 @@ const fieldDefs = {
58
58
  imageEndpoint: { type: 'string' },
59
59
  imageDefaultSize: { type: 'string' },
60
60
  transcriptionPrice: { type: 'number' },
61
+ embeddingPrice: { type: 'number' },
62
+ dimensions: { type: 'number' },
63
+ dimensionsSelectable: { type: 'boolean' },
64
+ maxBatch: { type: 'number' },
65
+ maxInputTokens: { type: 'number' },
66
+ inputTypes: { type: 'object' },
67
+ defaultInputType: { type: 'string' },
61
68
  deprecated: { type: 'string' },
62
69
  suspended: { type: 'string' },
63
70
  rpmLimit: { type: 'number' },
@@ -77,6 +84,7 @@ const COMPUTED_FIELDS = new Set(['upstreamIds'])
77
84
  const TYPE_CHECKERS = {
78
85
  string: (v) => typeof v === 'string',
79
86
  number: (v) => typeof v === 'number',
87
+ boolean: (v) => typeof v === 'boolean',
80
88
  array: (v) => Array.isArray(v),
81
89
  object: (v) => typeof v === 'object' && v !== null && !Array.isArray(v)
82
90
  }