mohdel 1.1.0 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,37 @@
1
+ /**
2
+ * Embedding-adapter registry. Mirrors `session/adapters/transcription` but
3
+ * scoped to providers with an embeddings endpoint.
4
+ *
5
+ * Five of mohdel's thirteen providers offer embeddings at all, and neither
6
+ * meta-provider does. Of those that do, only the base URL and the name of the
7
+ * dimension parameter differ across the OpenAI-shaped ones, so they share one
8
+ * adapter; Gemini and Cohere need their own.
9
+ *
10
+ * @module session/adapters/embedding
11
+ */
12
+
13
+ import { createEmbeddingAdapter } from './openai_compatible.js'
14
+ import { geminiEmbedding } from './gemini.js'
15
+ import { cohereEmbedding } from './cohere.js'
16
+
17
+ const EMBEDDING_ADAPTERS = {
18
+ openai: createEmbeddingAdapter({ baseURL: 'https://api.openai.com/v1' }),
19
+ // The endpoint is the catalog entry's `baseURL`, as it is for local chat.
20
+ local: createEmbeddingAdapter(),
21
+ gemini: geminiEmbedding,
22
+ cohere: cohereEmbedding
23
+ }
24
+
25
+ /** Providers with an embeddings adapter, for capability checks. */
26
+ export const EMBEDDING_PROVIDERS = Object.freeze(Object.keys(EMBEDDING_ADAPTERS))
27
+
28
+ /**
29
+ * @param {string} provider
30
+ */
31
+ export function getEmbeddingAdapter (provider) {
32
+ const adapter = EMBEDDING_ADAPTERS[provider]
33
+ if (!adapter) throw new Error(`no embedding adapter for provider: ${provider}`)
34
+ return adapter
35
+ }
36
+
37
+ export const embeddingAdapters = Object.freeze(EMBEDDING_ADAPTERS)
@@ -0,0 +1,92 @@
1
+ /**
2
+ * Shared embedding adapter for OpenAI-compatible `POST <baseURL>/embeddings`.
3
+ *
4
+ * Covers OpenAI and any self-hosted server that implements the endpoint
5
+ * (Ollama, vLLM, llama.cpp), and is the same shape Mistral, Fireworks and
6
+ * Qwen Cloud expose when they are added. Only the base URL and the name of the
7
+ * dimension parameter differ, so both are bound per provider in `./index.js`.
8
+ *
9
+ * @module session/adapters/embedding/openai_compatible
10
+ */
11
+
12
+ import { getSpec } from '../_catalog.js'
13
+ import { classifyProviderError, fromHttpStatus, typedError } from '../_errors.js'
14
+ import { computeEmbeddingCost } from '../_pricing.js'
15
+ import { catalogKey, bareOf } from '#core/model-id.js'
16
+ import { checkBatch, checkDimensions, resolveInputType, widthOf } from './_shared.js'
17
+
18
+ /**
19
+ * @param {{baseURL?: string, dimensionsField?: string}} config
20
+ */
21
+ export function createEmbeddingAdapter ({ baseURL, dimensionsField = 'dimensions' } = {}) {
22
+ return async function embedding (envelope, deps = {}) {
23
+ const fetchFn = deps.fetch ?? globalThis.fetch
24
+ const spec = deps.spec ?? getSpec(catalogKey(envelope.model)) ?? {}
25
+ const start = String(process.hrtime.bigint())
26
+
27
+ checkBatch(envelope, spec)
28
+ const dimensions = checkDimensions(envelope, spec)
29
+ const inputType = resolveInputType(envelope, spec)
30
+
31
+ // `local/` carries its endpoint on the entry; everything else is bound
32
+ // to a base URL by the registry.
33
+ const root = (spec.baseURL ?? baseURL ?? '').replace(/\/$/, '')
34
+ if (!root) {
35
+ throw typedError('no baseURL for this embedding model', 'CONFIGURATION_MISSING', false)
36
+ }
37
+
38
+ /** @type {Record<string, any>} */
39
+ const body = { model: spec.model ?? bareOf(envelope.model), input: envelope.input }
40
+ if (dimensions !== undefined) body[dimensionsField] = dimensions
41
+ if (inputType) body.input_type = inputType
42
+
43
+ let res
44
+ try {
45
+ res = await fetchFn(`${root}/embeddings`, {
46
+ method: 'POST',
47
+ headers: {
48
+ 'Content-Type': 'application/json',
49
+ ...(envelope.auth?.key ? { Authorization: `Bearer ${envelope.auth.key}` } : {})
50
+ },
51
+ body: JSON.stringify(body)
52
+ })
53
+ } catch (e) {
54
+ throw typedError(classifyProviderError(e, envelope.auth?.key).message, 'NET_ERROR', true)
55
+ }
56
+
57
+ if (!res.ok) {
58
+ const detail = await res.text().catch(() => '')
59
+ throw fromHttpStatus(res.status, detail, envelope.auth?.key)
60
+ }
61
+
62
+ const payload = await res.json()
63
+ const rows = Array.isArray(payload?.data) ? payload.data : []
64
+ // `index` is authoritative: the caller matches vectors to inputs by
65
+ // position, and a provider is free to answer out of order.
66
+ const vectors = rows
67
+ .slice()
68
+ .sort((a, b) => (a?.index ?? 0) - (b?.index ?? 0))
69
+ .map(r => r?.embedding)
70
+
71
+ if (vectors.length !== envelope.input.length || vectors.some(v => !Array.isArray(v))) {
72
+ throw typedError(
73
+ `expected ${envelope.input.length} vectors, got ${vectors.length}`,
74
+ 'EMBED_RESULT_MISMATCH',
75
+ false
76
+ )
77
+ }
78
+
79
+ const inputTokens = payload?.usage?.prompt_tokens ?? payload?.usage?.total_tokens ?? 0
80
+ const end = String(process.hrtime.bigint())
81
+
82
+ return {
83
+ status: 'completed',
84
+ vectors,
85
+ dimensions: widthOf(vectors),
86
+ inputType,
87
+ inputTokens,
88
+ cost: computeEmbeddingCost(spec, { inputTokens }),
89
+ timestamps: { start, first: end, end }
90
+ }
91
+ }
92
+ }
@@ -18,6 +18,7 @@ import { MAX_LINE_BYTES, exceedsLineBytes } from '#core/framing.js'
18
18
  import { run } from './run.js'
19
19
  import { runImage } from './run_image.js'
20
20
  import { runTranscription } from './run_transcription.js'
21
+ import { runEmbedding } from './run_embedding.js'
21
22
  import { setCatalog } from './adapters/_catalog.js'
22
23
 
23
24
  // Bounded memory for pre-dequeue cancels. Hostile/buggy supervisors
@@ -221,6 +222,16 @@ export async function drive (stdin, stdout) {
221
222
  } else {
222
223
  await writeLine({ type: 'error', error: out.error })
223
224
  }
225
+ } else if (envelope.op === 'embed') {
226
+ // Same one-shot contract; shape matches `js/core/embedding.js`
227
+ // after the tag strip. One line carries all N vectors.
228
+ const { op: _op, ...embEnv } = envelope
229
+ const out = await runEmbedding(embEnv)
230
+ if (out.ok) {
231
+ await writeLine({ type: 'embed_done', result: out.result })
232
+ } else {
233
+ await writeLine({ type: 'error', error: out.error })
234
+ }
224
235
  } else {
225
236
  for await (const ev of run(envelope, { signal: controller.signal })) {
226
237
  await writeLine(ev)
@@ -0,0 +1,94 @@
1
+ /**
2
+ * Embedding runtime. Resolves the adapter for the envelope's provider and
3
+ * returns either a result or a typed error, never throwing.
4
+ *
5
+ * Mirrors `run_transcription.js`: one synchronous request, no streaming, no
6
+ * cancellation path beyond the caller's own signal. Rate limits are enforced
7
+ * as in `run.js`, minus the speed lanes embeddings do not have.
8
+ *
9
+ * @module session/run_embedding
10
+ */
11
+
12
+ import { getEmbeddingAdapter } from './adapters/embedding/index.js'
13
+ import { classifyProviderError } from './adapters/_errors.js'
14
+ import { getProviderLimits } from './adapters/_providers.js'
15
+ import * as defaultLimiter from './_rate_limiter.js'
16
+ import { providerOf } from '#core/model-id.js'
17
+
18
+ /**
19
+ * @param {import('#core/embedding.js').EmbedEnvelope} envelope
20
+ * @param {{
21
+ * resolveAdapter?: (provider: string) => any,
22
+ * resolveProviderLimits?: (provider: string) => any,
23
+ * limiter?: any,
24
+ * sleep?: (ms: number) => Promise<void>,
25
+ * modelKey?: string,
26
+ * spec?: any
27
+ * }} [options]
28
+ * @returns {Promise<
29
+ * | {ok: true, result: import('#core/embedding.js').EmbedResult}
30
+ * | {ok: false, error: import('#core/errors.js').TypedError}
31
+ * >}
32
+ */
33
+ export async function runEmbedding (envelope, {
34
+ resolveAdapter = getEmbeddingAdapter,
35
+ resolveProviderLimits = getProviderLimits,
36
+ limiter = defaultLimiter,
37
+ sleep = defaultSleep,
38
+ modelKey = envelope.model,
39
+ spec
40
+ } = {}) {
41
+ const provider = providerOf(envelope.model)
42
+
43
+ let adapter
44
+ try {
45
+ adapter = resolveAdapter(provider)
46
+ } catch (e) {
47
+ return {
48
+ ok: false,
49
+ error: {
50
+ message: messageOf(e),
51
+ severity: 'error',
52
+ retryable: false,
53
+ type: 'SESSION_UNKNOWN_PROVIDER'
54
+ }
55
+ }
56
+ }
57
+
58
+ const providerCfg = resolveProviderLimits(provider) || {}
59
+ const rpmLimit = spec?.rpmLimit ?? providerCfg.rpmLimit
60
+ const tpmLimit = spec?.tpmLimit ?? providerCfg.tpmLimit
61
+ const inpmLimit = spec?.inpmLimit ?? providerCfg.inpmLimit
62
+ // `modelKey` is the catalog key, which the envelope carries over the wire but
63
+ // not on the in-process path, where it holds the upstream id instead.
64
+ const bucketKey = spec?.rateLimitScope === 'model' ? modelKey : provider
65
+ // A malformed envelope is metered as nothing: `checkBatch` rejects it inside
66
+ // the adapter, and a call that never reaches the provider must not spend quota.
67
+ const inputs = Array.isArray(envelope.input) ? envelope.input.length : 0
68
+
69
+ if (inputs > 0 && (rpmLimit != null || tpmLimit != null || inpmLimit != null)) {
70
+ const delay = limiter.check(bucketKey, { rpmLimit, tpmLimit, inpmLimit }, { inputs })
71
+ if (delay > 0) await sleep(delay)
72
+ limiter.recordRequest(bucketKey)
73
+ if (inpmLimit != null) limiter.recordInputs(bucketKey, inputs)
74
+ }
75
+
76
+ try {
77
+ const result = await adapter(envelope, spec ? { spec } : {})
78
+ if (tpmLimit != null && result.inputTokens) limiter.recordTokens(bucketKey, result.inputTokens)
79
+ return { ok: true, result }
80
+ } catch (e) {
81
+ const typed = /** @type {any} */(e).typed || classifyProviderError(e, envelope.auth?.key)
82
+ return { ok: false, error: typed }
83
+ }
84
+ }
85
+
86
+ /** @param {unknown} e */
87
+ function messageOf (e) {
88
+ return e instanceof Error ? e.message : String(e)
89
+ }
90
+
91
+ /** @param {number} ms */
92
+ function defaultSleep (ms) {
93
+ return new Promise(resolve => setTimeout(resolve, ms))
94
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mohdel",
3
- "version": "1.1.0",
3
+ "version": "1.3.0",
4
4
  "license": "MIT",
5
5
  "author": {
6
6
  "name": "Christophe Le Bars",
@@ -135,7 +135,7 @@
135
135
  "@opentelemetry/exporter-trace-otlp-grpc": "^0.222.0",
136
136
  "@opentelemetry/sdk-node": "^0.222.0",
137
137
  "chalk": "^6.0.0",
138
- "mohdel-thin-gate-linux-x64-gnu": "1.1.0"
138
+ "mohdel-thin-gate-linux-x64-gnu": "1.3.0"
139
139
  },
140
140
  "dependencies": {
141
141
  "@anthropic-ai/sdk": "^0.125.0",
@@ -38,6 +38,7 @@ const ARGUMENT = {
38
38
  'tag rm': ['model', 'tag'],
39
39
  'ratelimit show': ['model'],
40
40
  'ratelimit set': ['model'],
41
+ 'ratelimit provider set': ['provider'],
41
42
  'ratelimit rm': ['model']
42
43
  }
43
44
 
package/src/cli/index.js CHANGED
@@ -80,9 +80,9 @@ Commands:
80
80
  tag rm <model> <tag> Remove a tag
81
81
 
82
82
  ratelimit show <model|provider> Show effective limits (mo rl show)
83
- ratelimit set <model> [rpm] [tpm] Set model-level limits
83
+ ratelimit set <model> <limit> <value> Set limits: rpm, tpm, inpm
84
84
  ratelimit rm <model> Remove model-level limits
85
- ratelimit provider set <p> [rpm] [tpm] Set provider-level limits
85
+ ratelimit provider set <p> <limit> <v> Set provider-level limits
86
86
  ratelimit provider rm <p> Remove provider-level limits
87
87
 
88
88
  ask <provider/model> [prompt] One-shot inference (pipeable)
@@ -4,24 +4,111 @@ import { parseJsonFlag, jsonOutputOne } from './json-output.js'
4
4
  // CLI logger: silent for noisy levels, console.error for errors and fatals.
5
5
  const cliLogger = { ...silent, error: console.error, fatal: console.error }
6
6
 
7
+ const LIMIT_NAMES = ['rpm', 'tpm', 'inpm']
8
+
9
+ /**
10
+ * `0` is a killswitch, not "unset", so read nullability rather than truth.
11
+ *
12
+ * @param {{rpmLimit?: number, tpmLimit?: number, inpmLimit?: number} | null | undefined} entry
13
+ * @returns {string[]}
14
+ */
15
+ function limitParts (entry) {
16
+ if (!entry) return []
17
+ return LIMIT_NAMES
18
+ .filter(name => entry[`${name}Limit`] != null)
19
+ .map(name => `${name}=${entry[`${name}Limit`]}`)
20
+ }
21
+
22
+ /**
23
+ * @param {string[]} cleared Limits named on the command line; empty means all.
24
+ * @param {string[]} parts What is left afterwards.
25
+ */
26
+ function clearedLine (cleared, parts) {
27
+ if (cleared.length === 0) return 'limits cleared'
28
+ return `${cleared.join(', ')} cleared; ${parts.length ? `${parts.join(' ')} remain` : 'no limits remain'}`
29
+ }
30
+
31
+ /** @param {string[]} names */
32
+ function parseLimitNames (names) {
33
+ for (const name of names) {
34
+ if (!LIMIT_NAMES.includes(name)) {
35
+ console.error(`Unknown limit '${name}'. Known: ${LIMIT_NAMES.join(', ')}`)
36
+ process.exit(1)
37
+ }
38
+ }
39
+ return names
40
+ }
41
+
42
+ /** @param {string} raw */
43
+ function toCount (raw) {
44
+ const n = parseInt(raw, 10)
45
+ if (!Number.isInteger(n) || n < 0 || String(n) !== String(raw).trim()) {
46
+ console.error(`'${raw}' is not a whole number`)
47
+ process.exit(1)
48
+ }
49
+ return n
50
+ }
51
+
52
+ /**
53
+ * Two forms. Named pairs — `rpm 60 inpm 2000` — reach every limit, including
54
+ * one on its own. The positional `<rpm> [tpm]` covers the common pair; a
55
+ * leading digit picks that form, since no limit is named one.
56
+ *
57
+ * @param {string[]} args
58
+ * @param {string} usage
59
+ * @returns {{rpm?: number, tpm?: number, inpm?: number}}
60
+ */
61
+ function parseLimits (args, usage) {
62
+ if (args.length === 0) { console.error(usage); process.exit(1) }
63
+
64
+ if (/^\d/.test(args[0])) {
65
+ const [rpm, tpm] = args
66
+ return tpm ? { rpm: toCount(rpm), tpm: toCount(tpm) } : { rpm: toCount(rpm) }
67
+ }
68
+
69
+ /** @type {Record<string, number>} */
70
+ const limits = {}
71
+ for (let i = 0; i < args.length; i += 2) {
72
+ const name = args[i]
73
+ if (!LIMIT_NAMES.includes(name)) {
74
+ console.error(`Unknown limit '${name}'. Known: ${LIMIT_NAMES.join(', ')}`)
75
+ process.exit(1)
76
+ }
77
+ if (args[i + 1] == null) { console.error(`'${name}' needs a value`); process.exit(1) }
78
+ limits[name] = toCount(args[i + 1])
79
+ }
80
+ return limits
81
+ }
82
+
7
83
  export async function runRateLimit (args) {
8
84
  const jsonFlag = parseJsonFlag(args)
9
- const [action, arg1, arg2, arg3] = args
85
+ const [action, arg1] = args
10
86
 
11
87
  if (!action || action === '-h' || action === '--help') {
12
88
  console.log(`mohdel ratelimit — manage rate limits
13
89
 
14
90
  Usage:
15
91
  ratelimit show <model|provider> [--json] Show effective limits
16
- ratelimit set <model> [rpm] [tpm] Set model-level limits
17
- ratelimit rm <model> Remove model-level limits
18
- ratelimit provider set <provider> [rpm] [tpm] Set provider-level limits
19
- ratelimit provider rm <provider> Remove provider-level limits
92
+ ratelimit set <model> <limit> <value> … Set limits by name
93
+ ratelimit set <model> <rpm> [tpm] Shortcut for the two common ones
94
+ ratelimit rm <model> [limit …] Remove limits, or all of them
95
+ ratelimit provider set <provider> <limit> <value> …
96
+ ratelimit provider set <provider> <rpm> [tpm]
97
+ ratelimit provider rm <provider> [limit …] Remove limits, or all of them
98
+
99
+ Limits:
100
+ rpm requests per minute
101
+ tpm tokens per minute
102
+ inpm inputs per minute — what an embedding endpoint is metered in when
103
+ the provider counts inputs rather than requests or tokens
20
104
 
21
105
  Examples:
22
106
  ratelimit show anthropic Provider limits
23
107
  ratelimit show gemini/gemini-flash-latest Model limits, then provider
108
+ ratelimit set cohere/embed-v4.0 inpm 2000
109
+ ratelimit set gemini/gemini-flash-latest rpm 15 tpm 1000000
24
110
  ratelimit set gemini/gemini-flash-latest 15 1000000
111
+ ratelimit rm cohere/embed-v4.0 inpm
25
112
  ratelimit provider set anthropic 60 100000
26
113
 
27
114
  Aliases:
@@ -49,35 +136,24 @@ Configuration:
49
136
  if (providerAction === 'show') {
50
137
  if (!providerName) { console.error('Usage: ratelimit provider show <provider>'); process.exit(1) }
51
138
  const entry = mo.getProviderRateLimit(providerName)
52
- if (!entry) {
53
- console.log(`${providerName}: no limits set`)
54
- } else {
55
- const parts = []
56
- if (entry.rpmLimit) parts.push(`rpm=${entry.rpmLimit}`)
57
- if (entry.tpmLimit) parts.push(`tpm=${entry.tpmLimit}`)
58
- console.log(`${providerName}: ${parts.join(' ')}`)
59
- }
139
+ const parts = limitParts(entry)
140
+ console.log(parts.length ? `${providerName}: ${parts.join(' ')}` : `${providerName}: no limits set`)
60
141
  return
61
142
  }
62
143
 
63
144
  if (providerAction === 'set') {
64
145
  if (!providerName) { console.error('Usage: ratelimit provider set <provider> [rpm] [tpm]'); process.exit(1) }
65
- const [rpmStr, tpmStr] = providerArgs
66
- const rpm = rpmStr ? parseInt(rpmStr, 10) : undefined
67
- const tpm = tpmStr ? parseInt(tpmStr, 10) : undefined
68
- if (rpm == null && tpm == null) { console.error('Provide at least rpm or tpm'); process.exit(1) }
69
- const result = await mo.setProviderRateLimit(providerName, { rpm, tpm })
70
- const parts = []
71
- if (result.rpmLimit) parts.push(`rpm=${result.rpmLimit}`)
72
- if (result.tpmLimit) parts.push(`tpm=${result.tpmLimit}`)
73
- console.log(`${providerName}: ${parts.join(' ')}`)
146
+ const limits = parseLimits(providerArgs, 'Usage: ratelimit provider set <provider> <limit> <value> … | <rpm> [tpm]')
147
+ const result = await mo.setProviderRateLimit(providerName, limits)
148
+ console.log(`${providerName}: ${limitParts(result).join(' ')}`)
74
149
  return
75
150
  }
76
151
 
77
152
  if (providerAction === 'rm' || providerAction === 'remove') {
78
- if (!providerName) { console.error('Usage: ratelimit provider rm <provider>'); process.exit(1) }
79
- await mo.clearProviderRateLimit(providerName)
80
- console.log(`${providerName}: limits cleared`)
153
+ if (!providerName) { console.error('Usage: ratelimit provider rm <provider> [limit …]'); process.exit(1) }
154
+ const names = parseLimitNames(providerArgs)
155
+ const remaining = await mo.clearProviderRateLimit(providerName, names)
156
+ console.log(`${providerName}: ${clearedLine(names, limitParts(remaining))}`)
81
157
  return
82
158
  }
83
159
 
@@ -98,27 +174,24 @@ Configuration:
98
174
  const providerEntry = mo.getProviderRateLimit(info.provider) || {}
99
175
  const rpmLimit = info.rpmLimit ?? providerEntry.rpmLimit
100
176
  const tpmLimit = info.tpmLimit ?? providerEntry.tpmLimit
177
+ const inpmLimit = info.inpmLimit ?? providerEntry.inpmLimit
101
178
  const scope = info.rateLimitScope || 'provider'
102
- const source = (info.rpmLimit || info.tpmLimit) ? 'model' : 'provider'
179
+ const source = limitParts(info).length ? 'model' : 'provider'
103
180
  if (jsonFlag.json) {
104
- jsonOutputOne({ id: arg1, rpmLimit: rpmLimit || null, tpmLimit: tpmLimit || null, scope, source })
181
+ jsonOutputOne({ id: arg1, rpmLimit: rpmLimit || null, tpmLimit: tpmLimit || null, inpmLimit: inpmLimit || null, scope, source })
105
182
  return
106
183
  }
107
- if (!rpmLimit && !tpmLimit) {
184
+ const parts = limitParts({ rpmLimit, tpmLimit, inpmLimit })
185
+ if (parts.length === 0) {
108
186
  console.log(`${arg1}: no limits`)
109
187
  } else {
110
- const parts = []
111
- if (rpmLimit) parts.push(`rpm=${rpmLimit}`)
112
- if (tpmLimit) parts.push(`tpm=${tpmLimit}`)
113
- parts.push(`scope=${scope}`)
114
- parts.push(`(${source})`)
115
- console.log(`${arg1}: ${parts.join(' ')}`)
188
+ console.log(`${arg1}: ${[...parts, `scope=${scope}`, `(${source})`].join(' ')}`)
116
189
  }
117
190
  } else {
118
191
  // Treat as provider name
119
192
  const entry = mo.getProviderRateLimit(arg1)
120
193
  if (jsonFlag.json) {
121
- jsonOutputOne({ provider: arg1, rpmLimit: entry?.rpmLimit || null, tpmLimit: entry?.tpmLimit || null })
194
+ jsonOutputOne({ provider: arg1, rpmLimit: entry?.rpmLimit || null, tpmLimit: entry?.tpmLimit || null, inpmLimit: entry?.inpmLimit || null })
122
195
  return
123
196
  }
124
197
  if (!entry) {
@@ -127,6 +200,7 @@ Configuration:
127
200
  const parts = []
128
201
  if (entry.rpmLimit) parts.push(`rpm=${entry.rpmLimit}`)
129
202
  if (entry.tpmLimit) parts.push(`tpm=${entry.tpmLimit}`)
203
+ if (entry.inpmLimit) parts.push(`inpm=${entry.inpmLimit}`)
130
204
  console.log(`${arg1}: ${parts.join(' ')}`)
131
205
  }
132
206
  }
@@ -134,24 +208,21 @@ Configuration:
134
208
  }
135
209
 
136
210
  if (action === 'set') {
137
- if (!arg1) { console.error('Usage: ratelimit set <model> [rpm] [tpm]'); process.exit(1) }
138
- const rpm = arg2 ? parseInt(arg2, 10) : undefined
139
- const tpm = arg3 ? parseInt(arg3, 10) : undefined
140
- if (rpm == null && tpm == null) { console.error('Provide at least rpm or tpm'); process.exit(1) }
211
+ const usage = 'Usage: ratelimit set <model> <limit> <value> … | <rpm> [tpm]'
212
+ if (!arg1) { console.error(usage); process.exit(1) }
213
+ const limits = parseLimits(args.slice(2), usage)
141
214
  const model = useModel(arg1)
142
- const result = await model.setRateLimit({ rpm, tpm })
143
- const parts = []
144
- if (result.rpmLimit) parts.push(`rpm=${result.rpmLimit}`)
145
- if (result.tpmLimit) parts.push(`tpm=${result.tpmLimit}`)
146
- console.log(`${arg1}: ${parts.join(' ')} scope=model`)
215
+ const result = await model.setRateLimit(limits)
216
+ console.log(`${arg1}: ${limitParts(result).join(' ')} scope=model`)
147
217
  return
148
218
  }
149
219
 
150
220
  if (action === 'rm' || action === 'remove') {
151
- if (!arg1) { console.error('Usage: ratelimit rm <model>'); process.exit(1) }
221
+ if (!arg1) { console.error('Usage: ratelimit rm <model> [limit …]'); process.exit(1) }
222
+ const names = parseLimitNames(args.slice(2))
152
223
  const model = useModel(arg1)
153
- await model.clearRateLimit()
154
- console.log(`${arg1}: model limits cleared`)
224
+ const remaining = await model.clearRateLimit(names)
225
+ console.log(`${arg1}: ${clearedLine(names, limitParts(remaining))}`)
155
226
  return
156
227
  }
157
228
 
@@ -59,6 +59,18 @@ const creators = {
59
59
  logo: 'moonshotai.svg',
60
60
  description: 'Moonshot AI ships fluent, Chinese-first assistants and lean models tuned for consumer chat and business workflows.'
61
61
  },
62
+ cohere: {
63
+ prefixes: ['embed', 'command', 'rerank'],
64
+ label: 'Cohere',
65
+ logo: 'cohere.svg',
66
+ description: 'Cohere builds retrieval-focused models: embeddings and rerankers aimed at enterprise search rather than chat.'
67
+ },
68
+ nomic: {
69
+ prefixes: ['nomic-embed'],
70
+ label: 'Nomic',
71
+ logo: 'nomic.svg',
72
+ description: 'Nomic publishes open-weight embedding models with Matryoshka dimensions, widely self-hosted through Ollama and vLLM.'
73
+ },
62
74
  openai: {
63
75
  prefixes: ['gpt', 'whisper', 'dall-e', 'sora', 'text-embedding', 'o1', 'o3', 'o4'],
64
76
  label: 'OpenAI',
package/src/lib/index.js CHANGED
@@ -16,7 +16,7 @@ import {
16
16
  import { createRateLimiter } from '../../js/session/_rate_limiter.js'
17
17
  import { createCooldownTracker } from '../../js/session/_cooldown.js'
18
18
  import { setCatalog } from '../../js/session/adapters/_catalog.js'
19
- import { runAnswer, runAnswerImage, runAnswerTranscription } from '../../js/factory/bridge.js'
19
+ import { runAnswer, runAnswerEmbedding, runAnswerImage, runAnswerTranscription } from '../../js/factory/bridge.js'
20
20
  import { startSpan, endSpanOk, endSpanError } from './tracing.js'
21
21
  import { isValidTag } from './schema.js'
22
22
  import { silent } from './logger.js'
@@ -26,6 +26,8 @@ export const version = createRequire(import.meta.url)('../../package.json').vers
26
26
 
27
27
  const noop = () => {}
28
28
 
29
+ const LIMIT_FIELDS = ['rpmLimit', 'tpmLimit', 'inpmLimit']
30
+
29
31
  // Verbosity tiers — controls which mohdel internal log lines fire.
30
32
  //
31
33
  // 0 Anomaly-only. Failures, throttling, deprecation, server lifecycle.
@@ -413,30 +415,31 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
413
415
  return (providerName) => {
414
416
  const entry = providersConfig[providerName]
415
417
  if (!entry) return null
416
- const { rpmLimit, tpmLimit } = entry
417
- return (rpmLimit || tpmLimit) ? { rpmLimit, tpmLimit } : null
418
+ const { rpmLimit, tpmLimit, inpmLimit } = entry
419
+ return (rpmLimit || tpmLimit || inpmLimit) ? { rpmLimit, tpmLimit, inpmLimit } : null
418
420
  }
419
421
  }
420
422
 
421
423
  if (prop === 'setProviderRateLimit') {
422
- return async (providerName, { rpm, tpm } = {}) => {
424
+ return async (providerName, { rpm, tpm, inpm } = {}) => {
423
425
  const entry = providersConfig[providerName] || (providersConfig[providerName] = {})
424
426
  if (rpm != null) entry.rpmLimit = rpm
425
427
  if (tpm != null) entry.tpmLimit = tpm
428
+ if (inpm != null) entry.inpmLimit = inpm
426
429
  await saveProvidersConfig(providersConfig)
427
430
  return entry
428
431
  }
429
432
  }
430
433
 
431
434
  if (prop === 'clearProviderRateLimit') {
432
- return async (providerName) => {
435
+ return async (providerName, names = []) => {
433
436
  const entry = providersConfig[providerName]
434
- if (entry) {
435
- delete entry.rpmLimit
436
- delete entry.tpmLimit
437
- if (Object.keys(entry).length === 0) delete providersConfig[providerName]
438
- await saveProvidersConfig(providersConfig)
439
- }
437
+ if (!entry) return null
438
+ const fields = names.length ? names.map(n => `${n}Limit`) : LIMIT_FIELDS
439
+ for (const field of fields) delete entry[field]
440
+ if (Object.keys(entry).length === 0) delete providersConfig[providerName]
441
+ await saveProvidersConfig(providersConfig)
442
+ return providersConfig[providerName] || null
440
443
  }
441
444
  }
442
445
 
@@ -729,28 +732,44 @@ const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffor
729
732
  }
730
733
  }
731
734
 
735
+ if (prop === 'embed') {
736
+ return async (input, options = {}) => {
737
+ const { configuration } = await getRuntime()
738
+ return runAnswerEmbedding({
739
+ provider: modelSpec.provider,
740
+ model: modelSpec.model ?? resolvedModelId.split('/').pop(),
741
+ modelKey: resolvedModelId,
742
+ configuration,
743
+ input,
744
+ options,
745
+ spec: modelSpec
746
+ }, { limiter: rateLimiter, resolveProviderLimits })
747
+ }
748
+ }
749
+
732
750
  if (prop === 'setRateLimit') {
733
- return async ({ rpm, tpm } = {}) => {
751
+ return async ({ rpm, tpm, inpm } = {}) => {
734
752
  const curatedCache = getCuratedCacheSnapshot()
735
753
  const model = curatedCache[resolvedModelId] || (curatedCache[resolvedModelId] = { ...modelSpec })
736
754
  if (rpm != null) model.rpmLimit = rpm
737
755
  if (tpm != null) model.tpmLimit = tpm
756
+ if (inpm != null) model.inpmLimit = inpm
738
757
  model.rateLimitScope = 'model'
739
758
  await persistCuratedCache()
740
- return { rpmLimit: model.rpmLimit, tpmLimit: model.tpmLimit }
759
+ return { rpmLimit: model.rpmLimit, tpmLimit: model.tpmLimit, inpmLimit: model.inpmLimit }
741
760
  }
742
761
  }
743
762
 
744
763
  if (prop === 'clearRateLimit') {
745
- return async () => {
764
+ return async (names = []) => {
746
765
  const curatedCache = getCuratedCacheSnapshot()
747
766
  const model = curatedCache[resolvedModelId]
748
- if (model) {
749
- delete model.rpmLimit
750
- delete model.tpmLimit
751
- delete model.rateLimitScope
752
- await persistCuratedCache()
753
- }
767
+ if (!model) return {}
768
+ const fields = names.length ? names.map(n => `${n}Limit`) : LIMIT_FIELDS
769
+ for (const field of fields) delete model[field]
770
+ if (!LIMIT_FIELDS.some(field => model[field] != null)) delete model.rateLimitScope
771
+ await persistCuratedCache()
772
+ return { rpmLimit: model.rpmLimit, tpmLimit: model.tpmLimit, inpmLimit: model.inpmLimit }
754
773
  }
755
774
  }
756
775