mohdel 1.1.0 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/config/curated.schema.json +41 -3
- package/js/client/call_embedding.js +50 -0
- package/js/client/index.js +1 -0
- package/js/core/embedding.js +76 -0
- package/js/factory/bridge.js +44 -0
- package/js/session/_rate_limiter.js +33 -7
- package/js/session/adapters/_pricing.js +19 -1
- package/js/session/adapters/embedding/_shared.js +92 -0
- package/js/session/adapters/embedding/cohere.js +91 -0
- package/js/session/adapters/embedding/gemini.js +92 -0
- package/js/session/adapters/embedding/index.js +37 -0
- package/js/session/adapters/embedding/openai_compatible.js +92 -0
- package/js/session/driver.js +11 -0
- package/js/session/run_embedding.js +94 -0
- package/package.json +2 -2
- package/src/cli/complete.js +1 -0
- package/src/cli/index.js +2 -2
- package/src/cli/ratelimit.js +118 -47
- package/src/lib/creators.js +12 -0
- package/src/lib/index.js +39 -20
- package/src/lib/provider-info.js +7 -0
- package/src/lib/providers.js +12 -0
- package/src/lib/schema.js +9 -0
- package/src/lib/rate-limiter.js +0 -50
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Embedding-adapter registry. Mirrors `session/adapters/transcription` but
|
|
3
|
+
* scoped to providers with an embeddings endpoint.
|
|
4
|
+
*
|
|
5
|
+
* Five of mohdel's thirteen providers offer embeddings at all, and neither
|
|
6
|
+
* meta-provider does. Of those that do, only the base URL and the name of the
|
|
7
|
+
* dimension parameter differ across the OpenAI-shaped ones, so they share one
|
|
8
|
+
* adapter; Gemini and Cohere need their own.
|
|
9
|
+
*
|
|
10
|
+
* @module session/adapters/embedding
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { createEmbeddingAdapter } from './openai_compatible.js'
|
|
14
|
+
import { geminiEmbedding } from './gemini.js'
|
|
15
|
+
import { cohereEmbedding } from './cohere.js'
|
|
16
|
+
|
|
17
|
+
const EMBEDDING_ADAPTERS = {
|
|
18
|
+
openai: createEmbeddingAdapter({ baseURL: 'https://api.openai.com/v1' }),
|
|
19
|
+
// The endpoint is the catalog entry's `baseURL`, as it is for local chat.
|
|
20
|
+
local: createEmbeddingAdapter(),
|
|
21
|
+
gemini: geminiEmbedding,
|
|
22
|
+
cohere: cohereEmbedding
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
/** Providers with an embeddings adapter, for capability checks. */
|
|
26
|
+
export const EMBEDDING_PROVIDERS = Object.freeze(Object.keys(EMBEDDING_ADAPTERS))
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* @param {string} provider
|
|
30
|
+
*/
|
|
31
|
+
export function getEmbeddingAdapter (provider) {
|
|
32
|
+
const adapter = EMBEDDING_ADAPTERS[provider]
|
|
33
|
+
if (!adapter) throw new Error(`no embedding adapter for provider: ${provider}`)
|
|
34
|
+
return adapter
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export const embeddingAdapters = Object.freeze(EMBEDDING_ADAPTERS)
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared embedding adapter for OpenAI-compatible `POST <baseURL>/embeddings`.
|
|
3
|
+
*
|
|
4
|
+
* Covers OpenAI and any self-hosted server that implements the endpoint
|
|
5
|
+
* (Ollama, vLLM, llama.cpp), and is the same shape Mistral, Fireworks and
|
|
6
|
+
* Qwen Cloud expose when they are added. Only the base URL and the name of the
|
|
7
|
+
* dimension parameter differ, so both are bound per provider in `./index.js`.
|
|
8
|
+
*
|
|
9
|
+
* @module session/adapters/embedding/openai_compatible
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
import { getSpec } from '../_catalog.js'
|
|
13
|
+
import { classifyProviderError, fromHttpStatus, typedError } from '../_errors.js'
|
|
14
|
+
import { computeEmbeddingCost } from '../_pricing.js'
|
|
15
|
+
import { catalogKey, bareOf } from '#core/model-id.js'
|
|
16
|
+
import { checkBatch, checkDimensions, resolveInputType, widthOf } from './_shared.js'
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* @param {{baseURL?: string, dimensionsField?: string}} config
|
|
20
|
+
*/
|
|
21
|
+
export function createEmbeddingAdapter ({ baseURL, dimensionsField = 'dimensions' } = {}) {
|
|
22
|
+
return async function embedding (envelope, deps = {}) {
|
|
23
|
+
const fetchFn = deps.fetch ?? globalThis.fetch
|
|
24
|
+
const spec = deps.spec ?? getSpec(catalogKey(envelope.model)) ?? {}
|
|
25
|
+
const start = String(process.hrtime.bigint())
|
|
26
|
+
|
|
27
|
+
checkBatch(envelope, spec)
|
|
28
|
+
const dimensions = checkDimensions(envelope, spec)
|
|
29
|
+
const inputType = resolveInputType(envelope, spec)
|
|
30
|
+
|
|
31
|
+
// `local/` carries its endpoint on the entry; everything else is bound
|
|
32
|
+
// to a base URL by the registry.
|
|
33
|
+
const root = (spec.baseURL ?? baseURL ?? '').replace(/\/$/, '')
|
|
34
|
+
if (!root) {
|
|
35
|
+
throw typedError('no baseURL for this embedding model', 'CONFIGURATION_MISSING', false)
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/** @type {Record<string, any>} */
|
|
39
|
+
const body = { model: spec.model ?? bareOf(envelope.model), input: envelope.input }
|
|
40
|
+
if (dimensions !== undefined) body[dimensionsField] = dimensions
|
|
41
|
+
if (inputType) body.input_type = inputType
|
|
42
|
+
|
|
43
|
+
let res
|
|
44
|
+
try {
|
|
45
|
+
res = await fetchFn(`${root}/embeddings`, {
|
|
46
|
+
method: 'POST',
|
|
47
|
+
headers: {
|
|
48
|
+
'Content-Type': 'application/json',
|
|
49
|
+
...(envelope.auth?.key ? { Authorization: `Bearer ${envelope.auth.key}` } : {})
|
|
50
|
+
},
|
|
51
|
+
body: JSON.stringify(body)
|
|
52
|
+
})
|
|
53
|
+
} catch (e) {
|
|
54
|
+
throw typedError(classifyProviderError(e, envelope.auth?.key).message, 'NET_ERROR', true)
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
if (!res.ok) {
|
|
58
|
+
const detail = await res.text().catch(() => '')
|
|
59
|
+
throw fromHttpStatus(res.status, detail, envelope.auth?.key)
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
const payload = await res.json()
|
|
63
|
+
const rows = Array.isArray(payload?.data) ? payload.data : []
|
|
64
|
+
// `index` is authoritative: the caller matches vectors to inputs by
|
|
65
|
+
// position, and a provider is free to answer out of order.
|
|
66
|
+
const vectors = rows
|
|
67
|
+
.slice()
|
|
68
|
+
.sort((a, b) => (a?.index ?? 0) - (b?.index ?? 0))
|
|
69
|
+
.map(r => r?.embedding)
|
|
70
|
+
|
|
71
|
+
if (vectors.length !== envelope.input.length || vectors.some(v => !Array.isArray(v))) {
|
|
72
|
+
throw typedError(
|
|
73
|
+
`expected ${envelope.input.length} vectors, got ${vectors.length}`,
|
|
74
|
+
'EMBED_RESULT_MISMATCH',
|
|
75
|
+
false
|
|
76
|
+
)
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
const inputTokens = payload?.usage?.prompt_tokens ?? payload?.usage?.total_tokens ?? 0
|
|
80
|
+
const end = String(process.hrtime.bigint())
|
|
81
|
+
|
|
82
|
+
return {
|
|
83
|
+
status: 'completed',
|
|
84
|
+
vectors,
|
|
85
|
+
dimensions: widthOf(vectors),
|
|
86
|
+
inputType,
|
|
87
|
+
inputTokens,
|
|
88
|
+
cost: computeEmbeddingCost(spec, { inputTokens }),
|
|
89
|
+
timestamps: { start, first: end, end }
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
}
|
package/js/session/driver.js
CHANGED
|
@@ -18,6 +18,7 @@ import { MAX_LINE_BYTES, exceedsLineBytes } from '#core/framing.js'
|
|
|
18
18
|
import { run } from './run.js'
|
|
19
19
|
import { runImage } from './run_image.js'
|
|
20
20
|
import { runTranscription } from './run_transcription.js'
|
|
21
|
+
import { runEmbedding } from './run_embedding.js'
|
|
21
22
|
import { setCatalog } from './adapters/_catalog.js'
|
|
22
23
|
|
|
23
24
|
// Bounded memory for pre-dequeue cancels. Hostile/buggy supervisors
|
|
@@ -221,6 +222,16 @@ export async function drive (stdin, stdout) {
|
|
|
221
222
|
} else {
|
|
222
223
|
await writeLine({ type: 'error', error: out.error })
|
|
223
224
|
}
|
|
225
|
+
} else if (envelope.op === 'embed') {
|
|
226
|
+
// Same one-shot contract; shape matches `js/core/embedding.js`
|
|
227
|
+
// after the tag strip. One line carries all N vectors.
|
|
228
|
+
const { op: _op, ...embEnv } = envelope
|
|
229
|
+
const out = await runEmbedding(embEnv)
|
|
230
|
+
if (out.ok) {
|
|
231
|
+
await writeLine({ type: 'embed_done', result: out.result })
|
|
232
|
+
} else {
|
|
233
|
+
await writeLine({ type: 'error', error: out.error })
|
|
234
|
+
}
|
|
224
235
|
} else {
|
|
225
236
|
for await (const ev of run(envelope, { signal: controller.signal })) {
|
|
226
237
|
await writeLine(ev)
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Embedding runtime. Resolves the adapter for the envelope's provider and
|
|
3
|
+
* returns either a result or a typed error, never throwing.
|
|
4
|
+
*
|
|
5
|
+
* Mirrors `run_transcription.js`: one synchronous request, no streaming, no
|
|
6
|
+
* cancellation path beyond the caller's own signal. Rate limits are enforced
|
|
7
|
+
* as in `run.js`, minus the speed lanes embeddings do not have.
|
|
8
|
+
*
|
|
9
|
+
* @module session/run_embedding
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
import { getEmbeddingAdapter } from './adapters/embedding/index.js'
|
|
13
|
+
import { classifyProviderError } from './adapters/_errors.js'
|
|
14
|
+
import { getProviderLimits } from './adapters/_providers.js'
|
|
15
|
+
import * as defaultLimiter from './_rate_limiter.js'
|
|
16
|
+
import { providerOf } from '#core/model-id.js'
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* @param {import('#core/embedding.js').EmbedEnvelope} envelope
|
|
20
|
+
* @param {{
|
|
21
|
+
* resolveAdapter?: (provider: string) => any,
|
|
22
|
+
* resolveProviderLimits?: (provider: string) => any,
|
|
23
|
+
* limiter?: any,
|
|
24
|
+
* sleep?: (ms: number) => Promise<void>,
|
|
25
|
+
* modelKey?: string,
|
|
26
|
+
* spec?: any
|
|
27
|
+
* }} [options]
|
|
28
|
+
* @returns {Promise<
|
|
29
|
+
* | {ok: true, result: import('#core/embedding.js').EmbedResult}
|
|
30
|
+
* | {ok: false, error: import('#core/errors.js').TypedError}
|
|
31
|
+
* >}
|
|
32
|
+
*/
|
|
33
|
+
export async function runEmbedding (envelope, {
|
|
34
|
+
resolveAdapter = getEmbeddingAdapter,
|
|
35
|
+
resolveProviderLimits = getProviderLimits,
|
|
36
|
+
limiter = defaultLimiter,
|
|
37
|
+
sleep = defaultSleep,
|
|
38
|
+
modelKey = envelope.model,
|
|
39
|
+
spec
|
|
40
|
+
} = {}) {
|
|
41
|
+
const provider = providerOf(envelope.model)
|
|
42
|
+
|
|
43
|
+
let adapter
|
|
44
|
+
try {
|
|
45
|
+
adapter = resolveAdapter(provider)
|
|
46
|
+
} catch (e) {
|
|
47
|
+
return {
|
|
48
|
+
ok: false,
|
|
49
|
+
error: {
|
|
50
|
+
message: messageOf(e),
|
|
51
|
+
severity: 'error',
|
|
52
|
+
retryable: false,
|
|
53
|
+
type: 'SESSION_UNKNOWN_PROVIDER'
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
const providerCfg = resolveProviderLimits(provider) || {}
|
|
59
|
+
const rpmLimit = spec?.rpmLimit ?? providerCfg.rpmLimit
|
|
60
|
+
const tpmLimit = spec?.tpmLimit ?? providerCfg.tpmLimit
|
|
61
|
+
const inpmLimit = spec?.inpmLimit ?? providerCfg.inpmLimit
|
|
62
|
+
// `modelKey` is the catalog key, which the envelope carries over the wire but
|
|
63
|
+
// not on the in-process path, where it holds the upstream id instead.
|
|
64
|
+
const bucketKey = spec?.rateLimitScope === 'model' ? modelKey : provider
|
|
65
|
+
// A malformed envelope is metered as nothing: `checkBatch` rejects it inside
|
|
66
|
+
// the adapter, and a call that never reaches the provider must not spend quota.
|
|
67
|
+
const inputs = Array.isArray(envelope.input) ? envelope.input.length : 0
|
|
68
|
+
|
|
69
|
+
if (inputs > 0 && (rpmLimit != null || tpmLimit != null || inpmLimit != null)) {
|
|
70
|
+
const delay = limiter.check(bucketKey, { rpmLimit, tpmLimit, inpmLimit }, { inputs })
|
|
71
|
+
if (delay > 0) await sleep(delay)
|
|
72
|
+
limiter.recordRequest(bucketKey)
|
|
73
|
+
if (inpmLimit != null) limiter.recordInputs(bucketKey, inputs)
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
try {
|
|
77
|
+
const result = await adapter(envelope, spec ? { spec } : {})
|
|
78
|
+
if (tpmLimit != null && result.inputTokens) limiter.recordTokens(bucketKey, result.inputTokens)
|
|
79
|
+
return { ok: true, result }
|
|
80
|
+
} catch (e) {
|
|
81
|
+
const typed = /** @type {any} */(e).typed || classifyProviderError(e, envelope.auth?.key)
|
|
82
|
+
return { ok: false, error: typed }
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/** @param {unknown} e */
|
|
87
|
+
function messageOf (e) {
|
|
88
|
+
return e instanceof Error ? e.message : String(e)
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/** @param {number} ms */
|
|
92
|
+
function defaultSleep (ms) {
|
|
93
|
+
return new Promise(resolve => setTimeout(resolve, ms))
|
|
94
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "mohdel",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.3.0",
|
|
4
4
|
"license": "MIT",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "Christophe Le Bars",
|
|
@@ -135,7 +135,7 @@
|
|
|
135
135
|
"@opentelemetry/exporter-trace-otlp-grpc": "^0.222.0",
|
|
136
136
|
"@opentelemetry/sdk-node": "^0.222.0",
|
|
137
137
|
"chalk": "^6.0.0",
|
|
138
|
-
"mohdel-thin-gate-linux-x64-gnu": "1.
|
|
138
|
+
"mohdel-thin-gate-linux-x64-gnu": "1.3.0"
|
|
139
139
|
},
|
|
140
140
|
"dependencies": {
|
|
141
141
|
"@anthropic-ai/sdk": "^0.125.0",
|
package/src/cli/complete.js
CHANGED
package/src/cli/index.js
CHANGED
|
@@ -80,9 +80,9 @@ Commands:
|
|
|
80
80
|
tag rm <model> <tag> Remove a tag
|
|
81
81
|
|
|
82
82
|
ratelimit show <model|provider> Show effective limits (mo rl show)
|
|
83
|
-
ratelimit set <model>
|
|
83
|
+
ratelimit set <model> <limit> <value> Set limits: rpm, tpm, inpm
|
|
84
84
|
ratelimit rm <model> Remove model-level limits
|
|
85
|
-
ratelimit provider set <p>
|
|
85
|
+
ratelimit provider set <p> <limit> <v> Set provider-level limits
|
|
86
86
|
ratelimit provider rm <p> Remove provider-level limits
|
|
87
87
|
|
|
88
88
|
ask <provider/model> [prompt] One-shot inference (pipeable)
|
package/src/cli/ratelimit.js
CHANGED
|
@@ -4,24 +4,111 @@ import { parseJsonFlag, jsonOutputOne } from './json-output.js'
|
|
|
4
4
|
// CLI logger: silent for noisy levels, console.error for errors and fatals.
|
|
5
5
|
const cliLogger = { ...silent, error: console.error, fatal: console.error }
|
|
6
6
|
|
|
7
|
+
const LIMIT_NAMES = ['rpm', 'tpm', 'inpm']
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* `0` is a killswitch, not "unset", so read nullability rather than truth.
|
|
11
|
+
*
|
|
12
|
+
* @param {{rpmLimit?: number, tpmLimit?: number, inpmLimit?: number} | null | undefined} entry
|
|
13
|
+
* @returns {string[]}
|
|
14
|
+
*/
|
|
15
|
+
function limitParts (entry) {
|
|
16
|
+
if (!entry) return []
|
|
17
|
+
return LIMIT_NAMES
|
|
18
|
+
.filter(name => entry[`${name}Limit`] != null)
|
|
19
|
+
.map(name => `${name}=${entry[`${name}Limit`]}`)
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* @param {string[]} cleared Limits named on the command line; empty means all.
|
|
24
|
+
* @param {string[]} parts What is left afterwards.
|
|
25
|
+
*/
|
|
26
|
+
function clearedLine (cleared, parts) {
|
|
27
|
+
if (cleared.length === 0) return 'limits cleared'
|
|
28
|
+
return `${cleared.join(', ')} cleared; ${parts.length ? `${parts.join(' ')} remain` : 'no limits remain'}`
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/** @param {string[]} names */
|
|
32
|
+
function parseLimitNames (names) {
|
|
33
|
+
for (const name of names) {
|
|
34
|
+
if (!LIMIT_NAMES.includes(name)) {
|
|
35
|
+
console.error(`Unknown limit '${name}'. Known: ${LIMIT_NAMES.join(', ')}`)
|
|
36
|
+
process.exit(1)
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
return names
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
/** @param {string} raw */
|
|
43
|
+
function toCount (raw) {
|
|
44
|
+
const n = parseInt(raw, 10)
|
|
45
|
+
if (!Number.isInteger(n) || n < 0 || String(n) !== String(raw).trim()) {
|
|
46
|
+
console.error(`'${raw}' is not a whole number`)
|
|
47
|
+
process.exit(1)
|
|
48
|
+
}
|
|
49
|
+
return n
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Two forms. Named pairs — `rpm 60 inpm 2000` — reach every limit, including
|
|
54
|
+
* one on its own. The positional `<rpm> [tpm]` covers the common pair; a
|
|
55
|
+
* leading digit picks that form, since no limit is named one.
|
|
56
|
+
*
|
|
57
|
+
* @param {string[]} args
|
|
58
|
+
* @param {string} usage
|
|
59
|
+
* @returns {{rpm?: number, tpm?: number, inpm?: number}}
|
|
60
|
+
*/
|
|
61
|
+
function parseLimits (args, usage) {
|
|
62
|
+
if (args.length === 0) { console.error(usage); process.exit(1) }
|
|
63
|
+
|
|
64
|
+
if (/^\d/.test(args[0])) {
|
|
65
|
+
const [rpm, tpm] = args
|
|
66
|
+
return tpm ? { rpm: toCount(rpm), tpm: toCount(tpm) } : { rpm: toCount(rpm) }
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/** @type {Record<string, number>} */
|
|
70
|
+
const limits = {}
|
|
71
|
+
for (let i = 0; i < args.length; i += 2) {
|
|
72
|
+
const name = args[i]
|
|
73
|
+
if (!LIMIT_NAMES.includes(name)) {
|
|
74
|
+
console.error(`Unknown limit '${name}'. Known: ${LIMIT_NAMES.join(', ')}`)
|
|
75
|
+
process.exit(1)
|
|
76
|
+
}
|
|
77
|
+
if (args[i + 1] == null) { console.error(`'${name}' needs a value`); process.exit(1) }
|
|
78
|
+
limits[name] = toCount(args[i + 1])
|
|
79
|
+
}
|
|
80
|
+
return limits
|
|
81
|
+
}
|
|
82
|
+
|
|
7
83
|
export async function runRateLimit (args) {
|
|
8
84
|
const jsonFlag = parseJsonFlag(args)
|
|
9
|
-
const [action, arg1
|
|
85
|
+
const [action, arg1] = args
|
|
10
86
|
|
|
11
87
|
if (!action || action === '-h' || action === '--help') {
|
|
12
88
|
console.log(`mohdel ratelimit — manage rate limits
|
|
13
89
|
|
|
14
90
|
Usage:
|
|
15
91
|
ratelimit show <model|provider> [--json] Show effective limits
|
|
16
|
-
ratelimit set <model>
|
|
17
|
-
ratelimit
|
|
18
|
-
ratelimit
|
|
19
|
-
ratelimit provider
|
|
92
|
+
ratelimit set <model> <limit> <value> … Set limits by name
|
|
93
|
+
ratelimit set <model> <rpm> [tpm] Shortcut for the two common ones
|
|
94
|
+
ratelimit rm <model> [limit …] Remove limits, or all of them
|
|
95
|
+
ratelimit provider set <provider> <limit> <value> …
|
|
96
|
+
ratelimit provider set <provider> <rpm> [tpm]
|
|
97
|
+
ratelimit provider rm <provider> [limit …] Remove limits, or all of them
|
|
98
|
+
|
|
99
|
+
Limits:
|
|
100
|
+
rpm requests per minute
|
|
101
|
+
tpm tokens per minute
|
|
102
|
+
inpm inputs per minute — what an embedding endpoint is metered in when
|
|
103
|
+
the provider counts inputs rather than requests or tokens
|
|
20
104
|
|
|
21
105
|
Examples:
|
|
22
106
|
ratelimit show anthropic Provider limits
|
|
23
107
|
ratelimit show gemini/gemini-flash-latest Model limits, then provider
|
|
108
|
+
ratelimit set cohere/embed-v4.0 inpm 2000
|
|
109
|
+
ratelimit set gemini/gemini-flash-latest rpm 15 tpm 1000000
|
|
24
110
|
ratelimit set gemini/gemini-flash-latest 15 1000000
|
|
111
|
+
ratelimit rm cohere/embed-v4.0 inpm
|
|
25
112
|
ratelimit provider set anthropic 60 100000
|
|
26
113
|
|
|
27
114
|
Aliases:
|
|
@@ -49,35 +136,24 @@ Configuration:
|
|
|
49
136
|
if (providerAction === 'show') {
|
|
50
137
|
if (!providerName) { console.error('Usage: ratelimit provider show <provider>'); process.exit(1) }
|
|
51
138
|
const entry = mo.getProviderRateLimit(providerName)
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
} else {
|
|
55
|
-
const parts = []
|
|
56
|
-
if (entry.rpmLimit) parts.push(`rpm=${entry.rpmLimit}`)
|
|
57
|
-
if (entry.tpmLimit) parts.push(`tpm=${entry.tpmLimit}`)
|
|
58
|
-
console.log(`${providerName}: ${parts.join(' ')}`)
|
|
59
|
-
}
|
|
139
|
+
const parts = limitParts(entry)
|
|
140
|
+
console.log(parts.length ? `${providerName}: ${parts.join(' ')}` : `${providerName}: no limits set`)
|
|
60
141
|
return
|
|
61
142
|
}
|
|
62
143
|
|
|
63
144
|
if (providerAction === 'set') {
|
|
64
145
|
if (!providerName) { console.error('Usage: ratelimit provider set <provider> [rpm] [tpm]'); process.exit(1) }
|
|
65
|
-
const
|
|
66
|
-
const
|
|
67
|
-
|
|
68
|
-
if (rpm == null && tpm == null) { console.error('Provide at least rpm or tpm'); process.exit(1) }
|
|
69
|
-
const result = await mo.setProviderRateLimit(providerName, { rpm, tpm })
|
|
70
|
-
const parts = []
|
|
71
|
-
if (result.rpmLimit) parts.push(`rpm=${result.rpmLimit}`)
|
|
72
|
-
if (result.tpmLimit) parts.push(`tpm=${result.tpmLimit}`)
|
|
73
|
-
console.log(`${providerName}: ${parts.join(' ')}`)
|
|
146
|
+
const limits = parseLimits(providerArgs, 'Usage: ratelimit provider set <provider> <limit> <value> … | <rpm> [tpm]')
|
|
147
|
+
const result = await mo.setProviderRateLimit(providerName, limits)
|
|
148
|
+
console.log(`${providerName}: ${limitParts(result).join(' ')}`)
|
|
74
149
|
return
|
|
75
150
|
}
|
|
76
151
|
|
|
77
152
|
if (providerAction === 'rm' || providerAction === 'remove') {
|
|
78
|
-
if (!providerName) { console.error('Usage: ratelimit provider rm <provider>'); process.exit(1) }
|
|
79
|
-
|
|
80
|
-
|
|
153
|
+
if (!providerName) { console.error('Usage: ratelimit provider rm <provider> [limit …]'); process.exit(1) }
|
|
154
|
+
const names = parseLimitNames(providerArgs)
|
|
155
|
+
const remaining = await mo.clearProviderRateLimit(providerName, names)
|
|
156
|
+
console.log(`${providerName}: ${clearedLine(names, limitParts(remaining))}`)
|
|
81
157
|
return
|
|
82
158
|
}
|
|
83
159
|
|
|
@@ -98,27 +174,24 @@ Configuration:
|
|
|
98
174
|
const providerEntry = mo.getProviderRateLimit(info.provider) || {}
|
|
99
175
|
const rpmLimit = info.rpmLimit ?? providerEntry.rpmLimit
|
|
100
176
|
const tpmLimit = info.tpmLimit ?? providerEntry.tpmLimit
|
|
177
|
+
const inpmLimit = info.inpmLimit ?? providerEntry.inpmLimit
|
|
101
178
|
const scope = info.rateLimitScope || 'provider'
|
|
102
|
-
const source = (info.
|
|
179
|
+
const source = limitParts(info).length ? 'model' : 'provider'
|
|
103
180
|
if (jsonFlag.json) {
|
|
104
|
-
jsonOutputOne({ id: arg1, rpmLimit: rpmLimit || null, tpmLimit: tpmLimit || null, scope, source })
|
|
181
|
+
jsonOutputOne({ id: arg1, rpmLimit: rpmLimit || null, tpmLimit: tpmLimit || null, inpmLimit: inpmLimit || null, scope, source })
|
|
105
182
|
return
|
|
106
183
|
}
|
|
107
|
-
|
|
184
|
+
const parts = limitParts({ rpmLimit, tpmLimit, inpmLimit })
|
|
185
|
+
if (parts.length === 0) {
|
|
108
186
|
console.log(`${arg1}: no limits`)
|
|
109
187
|
} else {
|
|
110
|
-
|
|
111
|
-
if (rpmLimit) parts.push(`rpm=${rpmLimit}`)
|
|
112
|
-
if (tpmLimit) parts.push(`tpm=${tpmLimit}`)
|
|
113
|
-
parts.push(`scope=${scope}`)
|
|
114
|
-
parts.push(`(${source})`)
|
|
115
|
-
console.log(`${arg1}: ${parts.join(' ')}`)
|
|
188
|
+
console.log(`${arg1}: ${[...parts, `scope=${scope}`, `(${source})`].join(' ')}`)
|
|
116
189
|
}
|
|
117
190
|
} else {
|
|
118
191
|
// Treat as provider name
|
|
119
192
|
const entry = mo.getProviderRateLimit(arg1)
|
|
120
193
|
if (jsonFlag.json) {
|
|
121
|
-
jsonOutputOne({ provider: arg1, rpmLimit: entry?.rpmLimit || null, tpmLimit: entry?.tpmLimit || null })
|
|
194
|
+
jsonOutputOne({ provider: arg1, rpmLimit: entry?.rpmLimit || null, tpmLimit: entry?.tpmLimit || null, inpmLimit: entry?.inpmLimit || null })
|
|
122
195
|
return
|
|
123
196
|
}
|
|
124
197
|
if (!entry) {
|
|
@@ -127,6 +200,7 @@ Configuration:
|
|
|
127
200
|
const parts = []
|
|
128
201
|
if (entry.rpmLimit) parts.push(`rpm=${entry.rpmLimit}`)
|
|
129
202
|
if (entry.tpmLimit) parts.push(`tpm=${entry.tpmLimit}`)
|
|
203
|
+
if (entry.inpmLimit) parts.push(`inpm=${entry.inpmLimit}`)
|
|
130
204
|
console.log(`${arg1}: ${parts.join(' ')}`)
|
|
131
205
|
}
|
|
132
206
|
}
|
|
@@ -134,24 +208,21 @@ Configuration:
|
|
|
134
208
|
}
|
|
135
209
|
|
|
136
210
|
if (action === 'set') {
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
const
|
|
140
|
-
if (rpm == null && tpm == null) { console.error('Provide at least rpm or tpm'); process.exit(1) }
|
|
211
|
+
const usage = 'Usage: ratelimit set <model> <limit> <value> … | <rpm> [tpm]'
|
|
212
|
+
if (!arg1) { console.error(usage); process.exit(1) }
|
|
213
|
+
const limits = parseLimits(args.slice(2), usage)
|
|
141
214
|
const model = useModel(arg1)
|
|
142
|
-
const result = await model.setRateLimit(
|
|
143
|
-
|
|
144
|
-
if (result.rpmLimit) parts.push(`rpm=${result.rpmLimit}`)
|
|
145
|
-
if (result.tpmLimit) parts.push(`tpm=${result.tpmLimit}`)
|
|
146
|
-
console.log(`${arg1}: ${parts.join(' ')} scope=model`)
|
|
215
|
+
const result = await model.setRateLimit(limits)
|
|
216
|
+
console.log(`${arg1}: ${limitParts(result).join(' ')} scope=model`)
|
|
147
217
|
return
|
|
148
218
|
}
|
|
149
219
|
|
|
150
220
|
if (action === 'rm' || action === 'remove') {
|
|
151
|
-
if (!arg1) { console.error('Usage: ratelimit rm <model>'); process.exit(1) }
|
|
221
|
+
if (!arg1) { console.error('Usage: ratelimit rm <model> [limit …]'); process.exit(1) }
|
|
222
|
+
const names = parseLimitNames(args.slice(2))
|
|
152
223
|
const model = useModel(arg1)
|
|
153
|
-
await model.clearRateLimit()
|
|
154
|
-
console.log(`${arg1}:
|
|
224
|
+
const remaining = await model.clearRateLimit(names)
|
|
225
|
+
console.log(`${arg1}: ${clearedLine(names, limitParts(remaining))}`)
|
|
155
226
|
return
|
|
156
227
|
}
|
|
157
228
|
|
package/src/lib/creators.js
CHANGED
|
@@ -59,6 +59,18 @@ const creators = {
|
|
|
59
59
|
logo: 'moonshotai.svg',
|
|
60
60
|
description: 'Moonshot AI ships fluent, Chinese-first assistants and lean models tuned for consumer chat and business workflows.'
|
|
61
61
|
},
|
|
62
|
+
cohere: {
|
|
63
|
+
prefixes: ['embed', 'command', 'rerank'],
|
|
64
|
+
label: 'Cohere',
|
|
65
|
+
logo: 'cohere.svg',
|
|
66
|
+
description: 'Cohere builds retrieval-focused models: embeddings and rerankers aimed at enterprise search rather than chat.'
|
|
67
|
+
},
|
|
68
|
+
nomic: {
|
|
69
|
+
prefixes: ['nomic-embed'],
|
|
70
|
+
label: 'Nomic',
|
|
71
|
+
logo: 'nomic.svg',
|
|
72
|
+
description: 'Nomic publishes open-weight embedding models with Matryoshka dimensions, widely self-hosted through Ollama and vLLM.'
|
|
73
|
+
},
|
|
62
74
|
openai: {
|
|
63
75
|
prefixes: ['gpt', 'whisper', 'dall-e', 'sora', 'text-embedding', 'o1', 'o3', 'o4'],
|
|
64
76
|
label: 'OpenAI',
|
package/src/lib/index.js
CHANGED
|
@@ -16,7 +16,7 @@ import {
|
|
|
16
16
|
import { createRateLimiter } from '../../js/session/_rate_limiter.js'
|
|
17
17
|
import { createCooldownTracker } from '../../js/session/_cooldown.js'
|
|
18
18
|
import { setCatalog } from '../../js/session/adapters/_catalog.js'
|
|
19
|
-
import { runAnswer, runAnswerImage, runAnswerTranscription } from '../../js/factory/bridge.js'
|
|
19
|
+
import { runAnswer, runAnswerEmbedding, runAnswerImage, runAnswerTranscription } from '../../js/factory/bridge.js'
|
|
20
20
|
import { startSpan, endSpanOk, endSpanError } from './tracing.js'
|
|
21
21
|
import { isValidTag } from './schema.js'
|
|
22
22
|
import { silent } from './logger.js'
|
|
@@ -26,6 +26,8 @@ export const version = createRequire(import.meta.url)('../../package.json').vers
|
|
|
26
26
|
|
|
27
27
|
const noop = () => {}
|
|
28
28
|
|
|
29
|
+
const LIMIT_FIELDS = ['rpmLimit', 'tpmLimit', 'inpmLimit']
|
|
30
|
+
|
|
29
31
|
// Verbosity tiers — controls which mohdel internal log lines fire.
|
|
30
32
|
//
|
|
31
33
|
// 0 Anomaly-only. Failures, throttling, deprecation, server lifecycle.
|
|
@@ -413,30 +415,31 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
|
|
|
413
415
|
return (providerName) => {
|
|
414
416
|
const entry = providersConfig[providerName]
|
|
415
417
|
if (!entry) return null
|
|
416
|
-
const { rpmLimit, tpmLimit } = entry
|
|
417
|
-
return (rpmLimit || tpmLimit) ? { rpmLimit, tpmLimit } : null
|
|
418
|
+
const { rpmLimit, tpmLimit, inpmLimit } = entry
|
|
419
|
+
return (rpmLimit || tpmLimit || inpmLimit) ? { rpmLimit, tpmLimit, inpmLimit } : null
|
|
418
420
|
}
|
|
419
421
|
}
|
|
420
422
|
|
|
421
423
|
if (prop === 'setProviderRateLimit') {
|
|
422
|
-
return async (providerName, { rpm, tpm } = {}) => {
|
|
424
|
+
return async (providerName, { rpm, tpm, inpm } = {}) => {
|
|
423
425
|
const entry = providersConfig[providerName] || (providersConfig[providerName] = {})
|
|
424
426
|
if (rpm != null) entry.rpmLimit = rpm
|
|
425
427
|
if (tpm != null) entry.tpmLimit = tpm
|
|
428
|
+
if (inpm != null) entry.inpmLimit = inpm
|
|
426
429
|
await saveProvidersConfig(providersConfig)
|
|
427
430
|
return entry
|
|
428
431
|
}
|
|
429
432
|
}
|
|
430
433
|
|
|
431
434
|
if (prop === 'clearProviderRateLimit') {
|
|
432
|
-
return async (providerName) => {
|
|
435
|
+
return async (providerName, names = []) => {
|
|
433
436
|
const entry = providersConfig[providerName]
|
|
434
|
-
if (entry)
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
437
|
+
if (!entry) return null
|
|
438
|
+
const fields = names.length ? names.map(n => `${n}Limit`) : LIMIT_FIELDS
|
|
439
|
+
for (const field of fields) delete entry[field]
|
|
440
|
+
if (Object.keys(entry).length === 0) delete providersConfig[providerName]
|
|
441
|
+
await saveProvidersConfig(providersConfig)
|
|
442
|
+
return providersConfig[providerName] || null
|
|
440
443
|
}
|
|
441
444
|
}
|
|
442
445
|
|
|
@@ -729,28 +732,44 @@ const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffor
|
|
|
729
732
|
}
|
|
730
733
|
}
|
|
731
734
|
|
|
735
|
+
if (prop === 'embed') {
|
|
736
|
+
return async (input, options = {}) => {
|
|
737
|
+
const { configuration } = await getRuntime()
|
|
738
|
+
return runAnswerEmbedding({
|
|
739
|
+
provider: modelSpec.provider,
|
|
740
|
+
model: modelSpec.model ?? resolvedModelId.split('/').pop(),
|
|
741
|
+
modelKey: resolvedModelId,
|
|
742
|
+
configuration,
|
|
743
|
+
input,
|
|
744
|
+
options,
|
|
745
|
+
spec: modelSpec
|
|
746
|
+
}, { limiter: rateLimiter, resolveProviderLimits })
|
|
747
|
+
}
|
|
748
|
+
}
|
|
749
|
+
|
|
732
750
|
if (prop === 'setRateLimit') {
|
|
733
|
-
return async ({ rpm, tpm } = {}) => {
|
|
751
|
+
return async ({ rpm, tpm, inpm } = {}) => {
|
|
734
752
|
const curatedCache = getCuratedCacheSnapshot()
|
|
735
753
|
const model = curatedCache[resolvedModelId] || (curatedCache[resolvedModelId] = { ...modelSpec })
|
|
736
754
|
if (rpm != null) model.rpmLimit = rpm
|
|
737
755
|
if (tpm != null) model.tpmLimit = tpm
|
|
756
|
+
if (inpm != null) model.inpmLimit = inpm
|
|
738
757
|
model.rateLimitScope = 'model'
|
|
739
758
|
await persistCuratedCache()
|
|
740
|
-
return { rpmLimit: model.rpmLimit, tpmLimit: model.tpmLimit }
|
|
759
|
+
return { rpmLimit: model.rpmLimit, tpmLimit: model.tpmLimit, inpmLimit: model.inpmLimit }
|
|
741
760
|
}
|
|
742
761
|
}
|
|
743
762
|
|
|
744
763
|
if (prop === 'clearRateLimit') {
|
|
745
|
-
return async () => {
|
|
764
|
+
return async (names = []) => {
|
|
746
765
|
const curatedCache = getCuratedCacheSnapshot()
|
|
747
766
|
const model = curatedCache[resolvedModelId]
|
|
748
|
-
if (model) {
|
|
749
|
-
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
}
|
|
767
|
+
if (!model) return {}
|
|
768
|
+
const fields = names.length ? names.map(n => `${n}Limit`) : LIMIT_FIELDS
|
|
769
|
+
for (const field of fields) delete model[field]
|
|
770
|
+
if (!LIMIT_FIELDS.some(field => model[field] != null)) delete model.rateLimitScope
|
|
771
|
+
await persistCuratedCache()
|
|
772
|
+
return { rpmLimit: model.rpmLimit, tpmLimit: model.tpmLimit, inpmLimit: model.inpmLimit }
|
|
754
773
|
}
|
|
755
774
|
}
|
|
756
775
|
|