mohdel 1.1.0 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -422,6 +422,7 @@ OPENROUTER_API_SK=sk-or-...
422
422
  NOVITA_API_SK=...
423
423
  QWEN_API_SK=sk-...
424
424
  XIAOMI_API_SK=...
425
+ COHERE_API_SK=...
425
426
  MOHDEL_LOCAL_API_SK=...
426
427
  ```
427
428
 
@@ -462,6 +463,7 @@ What each provider supports through mohdel's unified interface:
462
463
  | Xiaomi | Yes | Yes | Yes | No | Auto | MiMo; shared chat-completions path, `reasoning_content` captured |
463
464
  | OpenRouter | Yes | Yes | Yes | No | Varies | Meta-provider; `providerOptions.openrouter` for routing prefs |
464
465
  | Local | Yes | Yes | Yes | No | No | Any OpenAI-compatible server; endpoint is the catalog entry's `baseURL` |
466
+ | Cohere | n/a | n/a | n/a | n/a | n/a | Embeddings only: no chat models reach mohdel through it |
465
467
  | Novita | Yes | Yes | Yes | No | Yes (`reasoning_content`) | Prices in the model list; text via the shared chat-completions path, separate image adapter |
466
468
 
467
469
  Adapter capability ≠ model capability — whether a given model accepts images, tools, or thinking effort depends on the model spec in `curated.json`. The adapter passes through what the envelope supplies; the provider rejects unsupported combos.
@@ -517,15 +517,53 @@
517
517
  "minimum": 0,
518
518
  "description": "USD per audio minute (transcription-type entries). Token-billed transcription models (OpenAI gpt-4o-*-transcribe) use inputPrice/outputPrice instead."
519
519
  },
520
+ "embeddingPrice": {
521
+ "type": "number",
522
+ "minimum": 0,
523
+ "description": "USD per 1,000,000 input tokens for an embedding model. Without it, cost is 0 on every embed call."
524
+ },
525
+ "dimensions": {
526
+ "type": "number",
527
+ "minimum": 1,
528
+ "description": "Native width of the vectors this model returns."
529
+ },
530
+ "dimensionsSelectable": {
531
+ "type": "boolean",
532
+ "description": "Whether the model accepts a requested output width. Providers spell the parameter differently (dimensions, output_dimension, outputDimensionality) and support it on some models only; when false, passing `dimensions` fails before dispatch instead of being ignored."
533
+ },
534
+ "maxBatch": {
535
+ "type": "number",
536
+ "minimum": 1,
537
+ "description": "Inputs accepted in one embed call. 2048 on OpenAI, 96 on Cohere, 10 on Qwen Cloud, 1 for Gemini's single-content endpoint. A larger batch fails before dispatch; mohdel does not split one, because splitting changes cost attribution and result ordering."
538
+ },
539
+ "maxInputTokens": {
540
+ "type": "number",
541
+ "minimum": 1,
542
+ "description": "Token ceiling for a single input, not for the request."
543
+ },
544
+ "inputTypes": {
545
+ "type": "object",
546
+ "additionalProperties": { "type": "string" },
547
+ "description": "Maps mohdel's symbolic roles (query, document, classification, clustering) to the provider's own vocabulary, e.g. {\"query\": \"RETRIEVAL_QUERY\"} for Gemini or {\"query\": \"search_query\"} for Cohere. An entry without it rejects `inputType` rather than dropping it: the role is baked into the vector, so a silent drop produces quietly worse retrieval."
548
+ },
549
+ "defaultInputType": {
550
+ "type": "string",
551
+ "description": "Symbolic role sent when the caller names none. Required for providers that make the parameter mandatory (Cohere v3+)."
552
+ },
520
553
  "rpmLimit": {
521
554
  "type": "integer",
522
555
  "minimum": 1,
523
- "description": "Requests per minute. Overrides provider default."
556
+ "description": "Requests per minute this key may send. Overrides the provider-level default."
524
557
  },
525
558
  "tpmLimit": {
526
559
  "type": "integer",
527
560
  "minimum": 1,
528
- "description": "Tokens per minute. Overrides provider default."
561
+ "description": "Tokens per minute this key may spend. Overrides the provider-level default."
562
+ },
563
+ "inpmLimit": {
564
+ "type": "integer",
565
+ "minimum": 1,
566
+ "description": "Inputs per minute this key may send to an embedding endpoint, for a provider that meters the endpoint in inputs rather than requests or tokens (Cohere publishes '2,000 inputs / min'). Counted exactly before dispatch from the batch size; a batch larger than the whole allowance is sent rather than delayed, since waiting cannot make it fit."
529
567
  },
530
568
  "rateLimitScope": {
531
569
  "type": "string",
@@ -533,7 +571,7 @@
533
571
  "model",
534
572
  "provider"
535
573
  ],
536
- "description": "'model' = private budget. 'provider' = shared with provider-level pool."
574
+ "description": "Whose budget this key's calls draw on: 'model' = a private bucket for this entry, 'provider' = the pool shared with every other model of the provider."
537
575
  },
538
576
  "deprecated": {
539
577
  "type": "string",
@@ -0,0 +1,50 @@
1
+ /**
2
+ * Send an EmbedEnvelope to thin-gate's `POST /v1/embed`.
3
+ *
4
+ * One-shot: a single JSON response body, no streaming, no cooldown or
5
+ * rate-limit. One call carries N inputs and returns N vectors in request
6
+ * order; the gate never splits a batch, so a batch larger than the model's
7
+ * `maxBatch` comes back as an error rather than as several calls.
8
+ *
9
+ * @module client/call_embedding
10
+ */
11
+
12
+ import { requestUnix } from './transport.js'
13
+ import { readAll, parseErrorBody } from './response.js'
14
+ import { MohdelError } from '#core'
15
+
16
+ /**
17
+ * @param {import('#core/embedding.js').EmbedEnvelope} envelope
18
+ * @param {object} options
19
+ * @param {string} options.socketPath
20
+ * @param {AbortSignal} [options.signal]
21
+ * @param {string} [options.path] HTTP path; defaults to '/v1/embed'
22
+ * @returns {Promise<import('#core/embedding.js').EmbedResult>}
23
+ */
24
+ export async function callEmbedding (envelope, { socketPath, signal, path = '/v1/embed' }) {
25
+ const res = await requestUnix({
26
+ socketPath,
27
+ path,
28
+ method: 'POST',
29
+ body: envelope,
30
+ signal
31
+ })
32
+
33
+ const body = await readAll(res)
34
+
35
+ if (res.statusCode !== 200) {
36
+ throw MohdelError.fromJSON(parseErrorBody(body, res.statusCode ?? 0))
37
+ }
38
+
39
+ let parsed
40
+ try {
41
+ parsed = JSON.parse(body)
42
+ } catch (e) {
43
+ throw new MohdelError(`gate returned an unparseable embed body: ${body.slice(0, 200)}`, {
44
+ type: 'PROTOCOL_INVALID_RESPONSE',
45
+ severity: 'error',
46
+ retryable: false
47
+ })
48
+ }
49
+ return parsed
50
+ }
@@ -16,4 +16,5 @@
16
16
  export { call } from './call.js'
17
17
  export { callImage } from './call_image.js'
18
18
  export { callTranscription } from './call_transcription.js'
19
+ export { callEmbedding } from './call_embedding.js'
19
20
  export { resolveGateBinary } from './gate-binary.js'
@@ -0,0 +1,76 @@
1
+ /**
2
+ * Embedding envelope and result.
3
+ *
4
+ * Separate call path from `CallEnvelope` / `AnswerResult`: embeddings are a
5
+ * single synchronous request/response, nothing is generated, and one call
6
+ * carries N inputs and returns N vectors in request order.
7
+ * Result shape: `{ status, vectors, dimensions, inputTokens, cost, timestamps }`.
8
+ *
9
+ * @module core/embedding
10
+ */
11
+
12
+ /**
13
+ * @typedef {object} EmbedEnvelope
14
+ *
15
+ * @property {string} callId
16
+ * @property {string} authId
17
+ * @property {import('./envelope.js').Auth} auth
18
+ * @property {string} [traceparent]
19
+ * @property {string} [baggage]
20
+ *
21
+ * @property {string} model
22
+ * Full mohdel id — `"<provider>/<bare>"`. Same shape as
23
+ * `CallEnvelope.model` (see `envelope.js`).
24
+ * @property {string[]} input
25
+ * Texts to embed. Always an array, even for one. A batch larger than the
26
+ * entry's `maxBatch` fails before dispatch rather than being split:
27
+ * splitting would change both cost attribution and result ordering.
28
+ *
29
+ * @property {number} [dimensions]
30
+ * Requested output width. Providers spell this three ways
31
+ * (`dimensions`, `output_dimension`, `outputDimensionality`) and support it
32
+ * on some models only; an entry without `dimensionsSelectable` rejects the
33
+ * field rather than sending it to be ignored.
34
+ * @property {string} [inputType]
35
+ * Symbolic role of the text: `query`, `document`, `classification`,
36
+ * `clustering`. Asymmetric models embed the same string differently
37
+ * depending on it, so a query and its matching passage land close together.
38
+ * The entry's `inputTypes` maps it to the provider's own vocabulary.
39
+ */
40
+
41
+ /**
42
+ * @typedef {object} EmbedResult
43
+ *
44
+ * @property {'completed'} status
45
+ * Embeddings are one-shot — no `incomplete` state.
46
+ * @property {number[][]} vectors
47
+ * One per input, in request order.
48
+ * @property {number} dimensions
49
+ * Width of the returned vectors, which is what the provider actually
50
+ * produced rather than what was asked for.
51
+ * @property {string | null} inputType
52
+ * The provider-native value that was sent, or null when none was. The role
53
+ * is baked into the vector, so a caller storing these must record it and
54
+ * query the same namespace the same way.
55
+ * @property {number} inputTokens
56
+ * @property {number} cost
57
+ * USD. `embeddingPrice` (per million input tokens) × `inputTokens`; 0 when
58
+ * the spec carries no price.
59
+ * @property {{start: string, first: string, end: string}} timestamps
60
+ * hrtime-bigint-as-string. `first` = `end` (no streaming).
61
+ */
62
+
63
+ export const EMBED_ENVELOPE_FIELDS = Object.freeze([
64
+ 'callId',
65
+ 'authId',
66
+ 'auth',
67
+ 'traceparent',
68
+ 'baggage',
69
+ 'model',
70
+ 'input',
71
+ 'dimensions',
72
+ 'inputType'
73
+ ])
74
+
75
+ /** Symbolic roles a caller may pass; the entry maps them to provider values. */
76
+ export const INPUT_TYPES = Object.freeze(['query', 'document', 'classification', 'clustering'])
@@ -23,6 +23,7 @@
23
23
  import { run } from '../session/run.js'
24
24
  import { runImage } from '../session/run_image.js'
25
25
  import { runTranscription } from '../session/run_transcription.js'
26
+ import { runEmbedding } from '../session/run_embedding.js'
26
27
  import { markTrustedMedia } from '../session/adapters/_media.js'
27
28
  import { MohdelError, validateIds } from '#core'
28
29
  import { createRealtimeDeltaBuffer } from '../../src/lib/utils.js'
@@ -180,6 +181,49 @@ export async function runAnswerTranscription ({ provider, model, configuration,
180
181
  return out.result
181
182
  }
182
183
 
184
+ /**
185
+ * Run an `embed()` call through the /session runtime.
186
+ *
187
+ * @param {object} args
188
+ * @param {string} args.provider
189
+ * @param {string} args.model
190
+ * @param {string} [args.modelKey] Mohdel catalog key, for the rate-limit
191
+ * bucket when the entry is model-scoped.
192
+ * @param {any} args.configuration
193
+ * @param {string | string[]} args.input One text or a batch; normalized to an
194
+ * array so the result shape never
195
+ * depends on how the caller asked.
196
+ * @param {any} [args.options] `inputType` / `dimensions` map onto
197
+ * the envelope; `callId` / `authId` are
198
+ * transport metadata.
199
+ * @param {any} [args.spec]
200
+ * @param {BridgeDeps} [deps]
201
+ * @returns {Promise<any>}
202
+ */
203
+ export async function runAnswerEmbedding ({ provider, model, modelKey, configuration, input, options = {}, spec }, deps = {}) {
204
+ const callId = options.callId || newCallId()
205
+ const authId = options.authId || 'local'
206
+ assertValidIds(callId, authId, `${provider}/${model}`)
207
+
208
+ const envelope = {
209
+ callId,
210
+ authId,
211
+ auth: configToAuth(configuration),
212
+ model: `${provider}/${model}`,
213
+ input: Array.isArray(input) ? input : [input]
214
+ }
215
+ if (options.inputType) envelope.inputType = options.inputType
216
+ if (options.dimensions !== undefined) envelope.dimensions = options.dimensions
217
+
218
+ const out = await runEmbedding(envelope, {
219
+ ...deps,
220
+ ...(modelKey ? { modelKey } : {}),
221
+ ...(spec ? { spec } : {})
222
+ })
223
+ if (!out.ok) throw MohdelError.fromJSON(out.error, { provider, model })
224
+ return out.result
225
+ }
226
+
183
227
  /**
184
228
  * @param {object} args
185
229
  * @param {string} args.modelKey Mohdel catalog key `<provider>/<bare>`. The
@@ -1,7 +1,9 @@
1
1
  /**
2
2
  * Minute-bucket rate limiter (per-key: provider or provider/model).
3
3
  *
4
- * Tracks RPM and TPM. Returns ms to wait if over limit — throttles
4
+ * Tracks RPM, TPM and INPM — inputs per minute, the unit an embedding
5
+ * endpoint is metered in when the provider counts inputs rather than
6
+ * requests or tokens. Returns ms to wait if over limit — throttles
5
7
  * rather than rejecting, so the caller can absorb small bursts
6
8
  * without a 429 round-trip.
7
9
  *
@@ -14,7 +16,7 @@
14
16
  */
15
17
 
16
18
  export function createRateLimiter () {
17
- /** @type {Map<string, {count: number, tokens: number, minute: number}>} */
19
+ /** @type {Map<string, {count: number, tokens: number, inputs: number, minute: number}>} */
18
20
  const buckets = new Map()
19
21
 
20
22
  const currentMinute = () => Math.floor(Date.now() / 60000)
@@ -24,7 +26,7 @@ export function createRateLimiter () {
24
26
  const minute = currentMinute()
25
27
  const b = buckets.get(key)
26
28
  if (b && b.minute === minute) return b
27
- const fresh = { count: 0, tokens: 0, minute }
29
+ const fresh = { count: 0, tokens: 0, inputs: 0, minute }
28
30
  buckets.set(key, fresh)
29
31
  return fresh
30
32
  }
@@ -42,15 +44,30 @@ export function createRateLimiter () {
42
44
  * returned regardless of the current bucket.
43
45
  * - positive number → throttle at that value.
44
46
  *
47
+ * `rpmLimit` and `tpmLimit` gate on what the bucket already holds:
48
+ * the size of the call ahead is not known until it returns.
49
+ * `inpmLimit` gates on what the call is about to add, which
50
+ * `pending.inputs` carries — a batch size is exact before dispatch,
51
+ * so the call is admitted only if the whole batch fits.
52
+ *
45
53
  * @param {string} key
46
- * @param {{rpmLimit?: number, tpmLimit?: number}} limits
54
+ * @param {{rpmLimit?: number, tpmLimit?: number, inpmLimit?: number}} limits
55
+ * @param {{inputs?: number}} [pending]
47
56
  * @returns {number}
48
57
  */
49
- const check = (key, { rpmLimit, tpmLimit } = {}) => {
50
- if (rpmLimit == null && tpmLimit == null) return 0
58
+ const check = (key, { rpmLimit, tpmLimit, inpmLimit } = {}, pending = {}) => {
59
+ if (rpmLimit == null && tpmLimit == null && inpmLimit == null) return 0
51
60
  const b = getBucket(key)
52
61
  if (rpmLimit != null && b.count >= rpmLimit) return msUntilNextMinute(b.minute)
53
62
  if (tpmLimit != null && b.tokens >= tpmLimit) return msUntilNextMinute(b.minute)
63
+ if (inpmLimit != null) {
64
+ const adding = pending.inputs ?? 0
65
+ if (inpmLimit === 0) return msUntilNextMinute(b.minute)
66
+ // A batch larger than the whole allowance never fits, so waiting out the
67
+ // minute buys nothing: send it and take the provider's answer. Splitting
68
+ // it is the caller's call, not mohdel's.
69
+ if (adding <= inpmLimit && b.inputs + adding > inpmLimit) return msUntilNextMinute(b.minute)
70
+ }
54
71
  return 0
55
72
  }
56
73
 
@@ -67,7 +84,15 @@ export function createRateLimiter () {
67
84
  getBucket(key).tokens += tokens
68
85
  }
69
86
 
70
- return { check, recordRequest, recordTokens }
87
+ /**
88
+ * @param {string} key
89
+ * @param {number} inputs
90
+ */
91
+ const recordInputs = (key, inputs) => {
92
+ getBucket(key).inputs += inputs
93
+ }
94
+
95
+ return { check, recordRequest, recordTokens, recordInputs }
71
96
  }
72
97
 
73
98
  // Single session-local instance.
@@ -75,3 +100,4 @@ const defaultLimiter = createRateLimiter()
75
100
  export const check = defaultLimiter.check
76
101
  export const recordRequest = defaultLimiter.recordRequest
77
102
  export const recordTokens = defaultLimiter.recordTokens
103
+ export const recordInputs = defaultLimiter.recordInputs
@@ -147,6 +147,21 @@ export function computeTranscriptionCost (spec, usage) {
147
147
  return 0
148
148
  }
149
149
 
150
+ /**
151
+ * Embedding cost. `embeddingPrice` is USD per million input tokens; nothing is
152
+ * generated, so there is no output side.
153
+ *
154
+ * @param {any} spec
155
+ * @param {{inputTokens?: number}} usage
156
+ * @returns {number}
157
+ */
158
+ export function computeEmbeddingCost (spec, usage) {
159
+ if (!spec || typeof spec.embeddingPrice !== 'number') return 0
160
+ const tokens = usage.inputTokens
161
+ if (typeof tokens !== 'number' || tokens <= 0) return 0
162
+ return round((tokens / 1_000_000) * spec.embeddingPrice)
163
+ }
164
+
150
165
  /**
151
166
  * Test convenience: inject pricing-only specs by model id. Wraps
152
167
  * `setCatalog` with the `{input, output, thinking?}` shape used in
@@ -168,6 +183,9 @@ export function setPricing (table) {
168
183
  }
169
184
 
170
185
  /** @param {number} n */
186
+ // 1e-10 USD, not 1e-6. Six decimals reads as plenty until you price a short
187
+ // embedding: 4 tokens at $0.02 per million is 8e-8, which rounds to zero, and
188
+ // a million such calls then sum to zero instead of to their real cost.
171
189
  function round (n) {
172
- return Math.round(n * 1e6) / 1e6
190
+ return Math.round(n * 1e10) / 1e10
173
191
  }
@@ -0,0 +1,92 @@
1
+ /**
2
+ * Pieces every embedding adapter needs: the pre-dispatch checks that turn a
3
+ * provider's per-model limits into an error before a request is sent, and the
4
+ * envelope-to-provider translation of `inputType`.
5
+ *
6
+ * These fail rather than drop. A `dimensions` the model ignores, or an
7
+ * `inputType` silently discarded, both produce vectors that look fine and
8
+ * retrieve badly, which is the failure mohdel exists to prevent.
9
+ *
10
+ * @module session/adapters/embedding/shared
11
+ */
12
+
13
+ import { typedError } from '../_errors.js'
14
+
15
+ /**
16
+ * @param {import('#core/embedding.js').EmbedEnvelope} envelope
17
+ * @param {any} spec
18
+ */
19
+ export function checkBatch (envelope, spec) {
20
+ const input = envelope.input
21
+ if (!Array.isArray(input) || input.length === 0) {
22
+ throw typedError('embed requires a non-empty input array', 'EMBED_INPUT_EMPTY', false)
23
+ }
24
+ if (input.some(t => typeof t !== 'string')) {
25
+ throw typedError('embed input must be strings', 'EMBED_INPUT_INVALID', false)
26
+ }
27
+ const max = spec?.maxBatch
28
+ if (typeof max === 'number' && input.length > max) {
29
+ throw typedError(
30
+ `${input.length} inputs exceeds this model's batch limit of ${max}`,
31
+ 'EMBED_BATCH_TOO_LARGE',
32
+ false
33
+ )
34
+ }
35
+ }
36
+
37
+ /**
38
+ * @param {import('#core/embedding.js').EmbedEnvelope} envelope
39
+ * @param {any} spec
40
+ * @returns {number | undefined}
41
+ */
42
+ export function checkDimensions (envelope, spec) {
43
+ const wanted = envelope.dimensions
44
+ if (wanted === undefined) return undefined
45
+ if (!spec?.dimensionsSelectable) {
46
+ throw typedError(
47
+ 'this model returns a fixed vector width; remove `dimensions`',
48
+ 'EMBED_DIMENSIONS_UNSUPPORTED',
49
+ false
50
+ )
51
+ }
52
+ return wanted
53
+ }
54
+
55
+ /**
56
+ * Symbolic role to the provider's own vocabulary. The entry owns the mapping,
57
+ * the same way `thinkingEffortLevels` owns thinking budgets.
58
+ *
59
+ * @param {import('#core/embedding.js').EmbedEnvelope} envelope
60
+ * @param {any} spec
61
+ * @returns {string | null}
62
+ */
63
+ export function resolveInputType (envelope, spec) {
64
+ const table = spec?.inputTypes
65
+ const wanted = envelope.inputType ?? spec?.defaultInputType ?? null
66
+ if (wanted === null) return null
67
+
68
+ if (!table || typeof table !== 'object') {
69
+ throw typedError(
70
+ 'this model declares no inputTypes; remove `inputType`',
71
+ 'EMBED_INPUT_TYPE_UNSUPPORTED',
72
+ false
73
+ )
74
+ }
75
+ const native = table[wanted]
76
+ if (typeof native !== 'string') {
77
+ throw typedError(
78
+ `inputType '${wanted}' is not declared for this model; known: ${Object.keys(table).join(', ')}`,
79
+ 'EMBED_INPUT_TYPE_UNKNOWN',
80
+ false
81
+ )
82
+ }
83
+ return native
84
+ }
85
+
86
+ /**
87
+ * @param {number[][]} vectors
88
+ * @returns {number}
89
+ */
90
+ export function widthOf (vectors) {
91
+ return vectors.length && Array.isArray(vectors[0]) ? vectors[0].length : 0
92
+ }
@@ -0,0 +1,91 @@
1
+ /**
2
+ * Cohere embedding adapter.
3
+ *
4
+ * The most divergent shape mohdel talks to: `POST /v2/embed` rather than
5
+ * `/embeddings`, inputs under `texts`, `input_type` **required**, and
6
+ * `embeddings` returned as an object keyed by dtype
7
+ * (`{ float: [[...]] }`) instead of a `data[]` array. Tokens live at
8
+ * `meta.billed_units.input_tokens`.
9
+ *
10
+ * @module session/adapters/embedding/cohere
11
+ */
12
+
13
+ import { getSpec } from '../_catalog.js'
14
+ import { classifyProviderError, fromHttpStatus, typedError } from '../_errors.js'
15
+ import { computeEmbeddingCost } from '../_pricing.js'
16
+ import { catalogKey, bareOf } from '#core/model-id.js'
17
+ import { checkBatch, checkDimensions, resolveInputType, widthOf } from './_shared.js'
18
+
19
+ const BASE_URL = 'https://api.cohere.com/v2'
20
+
21
+ export async function cohereEmbedding (envelope, deps = {}) {
22
+ const fetchFn = deps.fetch ?? globalThis.fetch
23
+ const spec = deps.spec ?? getSpec(catalogKey(envelope.model)) ?? {}
24
+ const start = String(process.hrtime.bigint())
25
+
26
+ checkBatch(envelope, spec)
27
+ const dimensions = checkDimensions(envelope, spec)
28
+ const inputType = resolveInputType(envelope, spec)
29
+
30
+ // Cohere rejects a call without it, so an entry that declares no mapping is
31
+ // unusable rather than merely less accurate.
32
+ if (!inputType) {
33
+ throw typedError(
34
+ 'cohere requires an input_type; set inputType or the entry\'s defaultInputType',
35
+ 'EMBED_INPUT_TYPE_REQUIRED',
36
+ false
37
+ )
38
+ }
39
+
40
+ const body = {
41
+ model: spec.model ?? bareOf(envelope.model),
42
+ texts: envelope.input,
43
+ input_type: inputType,
44
+ embedding_types: ['float'],
45
+ ...(dimensions !== undefined ? { output_dimension: dimensions } : {})
46
+ }
47
+
48
+ let res
49
+ try {
50
+ res = await fetchFn(`${BASE_URL}/embed`, {
51
+ method: 'POST',
52
+ headers: {
53
+ 'Content-Type': 'application/json',
54
+ Authorization: `Bearer ${envelope.auth.key}`
55
+ },
56
+ body: JSON.stringify(body)
57
+ })
58
+ } catch (e) {
59
+ throw typedError(classifyProviderError(e, envelope.auth?.key).message, 'NET_ERROR', true)
60
+ }
61
+
62
+ if (!res.ok) {
63
+ const detail = await res.text().catch(() => '')
64
+ throw fromHttpStatus(res.status, detail, envelope.auth?.key)
65
+ }
66
+
67
+ const payload = await res.json()
68
+ const vectors = payload?.embeddings?.float ?? []
69
+
70
+ if (vectors.length !== envelope.input.length || vectors.some(v => !Array.isArray(v))) {
71
+ throw typedError(
72
+ `expected ${envelope.input.length} vectors, got ${vectors.length}`,
73
+ 'EMBED_RESULT_MISMATCH',
74
+ false
75
+ )
76
+ }
77
+
78
+ const inputTokens = payload?.meta?.billed_units?.input_tokens ??
79
+ payload?.meta?.tokens?.input_tokens ?? 0
80
+ const end = String(process.hrtime.bigint())
81
+
82
+ return {
83
+ status: 'completed',
84
+ vectors,
85
+ dimensions: widthOf(vectors),
86
+ inputType,
87
+ inputTokens,
88
+ cost: computeEmbeddingCost(spec, { inputTokens }),
89
+ timestamps: { start, first: end, end }
90
+ }
91
+ }
@@ -0,0 +1,92 @@
1
+ /**
2
+ * Gemini embedding adapter.
3
+ *
4
+ * Shares nothing with the OpenAI shape: a different endpoint per batch size
5
+ * (`:embedContent` for one input, `:batchEmbedContents` for several), vectors
6
+ * under `embedding.values`, and tokens under
7
+ * `usageMetadata.promptTokenCount`. The dimension parameter is
8
+ * `outputDimensionality` and lives inside a config object rather than at the
9
+ * top level.
10
+ *
11
+ * @module session/adapters/embedding/gemini
12
+ */
13
+
14
+ import { getSpec } from '../_catalog.js'
15
+ import { classifyProviderError, fromHttpStatus, typedError } from '../_errors.js'
16
+ import { computeEmbeddingCost } from '../_pricing.js'
17
+ import { catalogKey, bareOf } from '#core/model-id.js'
18
+ import { checkBatch, checkDimensions, resolveInputType, widthOf } from './_shared.js'
19
+
20
+ const BASE_URL = 'https://generativelanguage.googleapis.com/v1beta'
21
+
22
+ export async function geminiEmbedding (envelope, deps = {}) {
23
+ const fetchFn = deps.fetch ?? globalThis.fetch
24
+ const spec = deps.spec ?? getSpec(catalogKey(envelope.model)) ?? {}
25
+ const start = String(process.hrtime.bigint())
26
+
27
+ checkBatch(envelope, spec)
28
+ const dimensions = checkDimensions(envelope, spec)
29
+ const taskType = resolveInputType(envelope, spec)
30
+
31
+ const model = spec.model ?? bareOf(envelope.model)
32
+ const name = model.startsWith('models/') ? model : `models/${model}`
33
+ const batched = envelope.input.length > 1
34
+
35
+ const one = (text) => ({
36
+ model: name,
37
+ content: { parts: [{ text }] },
38
+ ...(taskType ? { taskType } : {}),
39
+ ...(dimensions !== undefined ? { outputDimensionality: dimensions } : {})
40
+ })
41
+ const body = batched
42
+ ? { requests: envelope.input.map(one) }
43
+ : one(envelope.input[0])
44
+
45
+ const method = batched ? 'batchEmbedContents' : 'embedContent'
46
+ let res
47
+ try {
48
+ res = await fetchFn(`${BASE_URL}/${name}:${method}`, {
49
+ method: 'POST',
50
+ headers: {
51
+ 'Content-Type': 'application/json',
52
+ 'x-goog-api-key': envelope.auth.key
53
+ },
54
+ body: JSON.stringify(body)
55
+ })
56
+ } catch (e) {
57
+ throw typedError(classifyProviderError(e, envelope.auth?.key).message, 'NET_ERROR', true)
58
+ }
59
+
60
+ if (!res.ok) {
61
+ const detail = await res.text().catch(() => '')
62
+ throw fromHttpStatus(res.status, detail, envelope.auth?.key)
63
+ }
64
+
65
+ const payload = await res.json()
66
+ const vectors = batched
67
+ ? (payload?.embeddings ?? []).map(e => e?.values)
68
+ : [payload?.embedding?.values]
69
+
70
+ if (vectors.length !== envelope.input.length || vectors.some(v => !Array.isArray(v))) {
71
+ throw typedError(
72
+ `expected ${envelope.input.length} vectors, got ${vectors.length}`,
73
+ 'EMBED_RESULT_MISMATCH',
74
+ false
75
+ )
76
+ }
77
+
78
+ // Only the single-content endpoint reports usage; the batch one does not,
79
+ // so a batched call prices at 0 rather than on a guessed token count.
80
+ const inputTokens = payload?.usageMetadata?.promptTokenCount ?? 0
81
+ const end = String(process.hrtime.bigint())
82
+
83
+ return {
84
+ status: 'completed',
85
+ vectors,
86
+ dimensions: widthOf(vectors),
87
+ inputType: taskType,
88
+ inputTokens,
89
+ cost: computeEmbeddingCost(spec, { inputTokens }),
90
+ timestamps: { start, first: end, end }
91
+ }
92
+ }