mohdel 1.2.0 → 1.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -553,12 +553,17 @@
553
553
  "rpmLimit": {
554
554
  "type": "integer",
555
555
  "minimum": 1,
556
- "description": "Requests per minute. Overrides provider default."
556
+ "description": "Requests per minute this key may send. Overrides the provider-level default."
557
557
  },
558
558
  "tpmLimit": {
559
559
  "type": "integer",
560
560
  "minimum": 1,
561
- "description": "Tokens per minute. Overrides provider default."
561
+ "description": "Tokens per minute this key may spend. Overrides the provider-level default."
562
+ },
563
+ "inpmLimit": {
564
+ "type": "integer",
565
+ "minimum": 1,
566
+ "description": "Inputs per minute this key may send to an embedding endpoint, for a provider that meters the endpoint in inputs rather than requests or tokens (Cohere publishes '2,000 inputs / min'). Counted exactly before dispatch from the batch size; a batch larger than the whole allowance is sent rather than delayed, since waiting cannot make it fit."
562
567
  },
563
568
  "rateLimitScope": {
564
569
  "type": "string",
@@ -566,7 +571,7 @@
566
571
  "model",
567
572
  "provider"
568
573
  ],
569
- "description": "'model' = private budget. 'provider' = shared with provider-level pool."
574
+ "description": "Whose budget this key's calls draw on: 'model' = a private bucket for this entry, 'provider' = the pool shared with every other model of the provider."
570
575
  },
571
576
  "deprecated": {
572
577
  "type": "string",
@@ -187,6 +187,8 @@ export async function runAnswerTranscription ({ provider, model, configuration,
187
187
  * @param {object} args
188
188
  * @param {string} args.provider
189
189
  * @param {string} args.model
190
+ * @param {string} [args.modelKey] Mohdel catalog key, for the rate-limit
191
+ * bucket when the entry is model-scoped.
190
192
  * @param {any} args.configuration
191
193
  * @param {string | string[]} args.input One text or a batch; normalized to an
192
194
  * array so the result shape never
@@ -195,9 +197,10 @@ export async function runAnswerTranscription ({ provider, model, configuration,
195
197
  * the envelope; `callId` / `authId` are
196
198
  * transport metadata.
197
199
  * @param {any} [args.spec]
200
+ * @param {BridgeDeps} [deps]
198
201
  * @returns {Promise<any>}
199
202
  */
200
- export async function runAnswerEmbedding ({ provider, model, configuration, input, options = {}, spec }) {
203
+ export async function runAnswerEmbedding ({ provider, model, modelKey, configuration, input, options = {}, spec }, deps = {}) {
201
204
  const callId = options.callId || newCallId()
202
205
  const authId = options.authId || 'local'
203
206
  assertValidIds(callId, authId, `${provider}/${model}`)
@@ -212,7 +215,11 @@ export async function runAnswerEmbedding ({ provider, model, configuration, inpu
212
215
  if (options.inputType) envelope.inputType = options.inputType
213
216
  if (options.dimensions !== undefined) envelope.dimensions = options.dimensions
214
217
 
215
- const out = await runEmbedding(envelope, spec ? { spec } : {})
218
+ const out = await runEmbedding(envelope, {
219
+ ...deps,
220
+ ...(modelKey ? { modelKey } : {}),
221
+ ...(spec ? { spec } : {})
222
+ })
216
223
  if (!out.ok) throw MohdelError.fromJSON(out.error, { provider, model })
217
224
  return out.result
218
225
  }
@@ -1,7 +1,9 @@
1
1
  /**
2
2
  * Minute-bucket rate limiter (per-key: provider or provider/model).
3
3
  *
4
- * Tracks RPM and TPM. Returns ms to wait if over limit — throttles
4
+ * Tracks RPM, TPM and INPM — inputs per minute, the unit an embedding
5
+ * endpoint is metered in when the provider counts inputs rather than
6
+ * requests or tokens. Returns ms to wait if over limit — throttles
5
7
  * rather than rejecting, so the caller can absorb small bursts
6
8
  * without a 429 round-trip.
7
9
  *
@@ -14,7 +16,7 @@
14
16
  */
15
17
 
16
18
  export function createRateLimiter () {
17
- /** @type {Map<string, {count: number, tokens: number, minute: number}>} */
19
+ /** @type {Map<string, {count: number, tokens: number, inputs: number, minute: number}>} */
18
20
  const buckets = new Map()
19
21
 
20
22
  const currentMinute = () => Math.floor(Date.now() / 60000)
@@ -24,7 +26,7 @@ export function createRateLimiter () {
24
26
  const minute = currentMinute()
25
27
  const b = buckets.get(key)
26
28
  if (b && b.minute === minute) return b
27
- const fresh = { count: 0, tokens: 0, minute }
29
+ const fresh = { count: 0, tokens: 0, inputs: 0, minute }
28
30
  buckets.set(key, fresh)
29
31
  return fresh
30
32
  }
@@ -42,15 +44,30 @@ export function createRateLimiter () {
42
44
  * returned regardless of the current bucket.
43
45
  * - positive number → throttle at that value.
44
46
  *
47
+ * `rpmLimit` and `tpmLimit` gate on what the bucket already holds:
48
+ * the size of the call ahead is not known until it returns.
49
+ * `inpmLimit` gates on what the call is about to add, which
50
+ * `pending.inputs` carries — a batch size is exact before dispatch,
51
+ * so the call is admitted only if the whole batch fits.
52
+ *
45
53
  * @param {string} key
46
- * @param {{rpmLimit?: number, tpmLimit?: number}} limits
54
+ * @param {{rpmLimit?: number, tpmLimit?: number, inpmLimit?: number}} limits
55
+ * @param {{inputs?: number}} [pending]
47
56
  * @returns {number}
48
57
  */
49
- const check = (key, { rpmLimit, tpmLimit } = {}) => {
50
- if (rpmLimit == null && tpmLimit == null) return 0
58
+ const check = (key, { rpmLimit, tpmLimit, inpmLimit } = {}, pending = {}) => {
59
+ if (rpmLimit == null && tpmLimit == null && inpmLimit == null) return 0
51
60
  const b = getBucket(key)
52
61
  if (rpmLimit != null && b.count >= rpmLimit) return msUntilNextMinute(b.minute)
53
62
  if (tpmLimit != null && b.tokens >= tpmLimit) return msUntilNextMinute(b.minute)
63
+ if (inpmLimit != null) {
64
+ const adding = pending.inputs ?? 0
65
+ if (inpmLimit === 0) return msUntilNextMinute(b.minute)
66
+ // A batch larger than the whole allowance never fits, so waiting out the
67
+ // minute buys nothing: send it and take the provider's answer. Splitting
68
+ // it is the caller's call, not mohdel's.
69
+ if (adding <= inpmLimit && b.inputs + adding > inpmLimit) return msUntilNextMinute(b.minute)
70
+ }
54
71
  return 0
55
72
  }
56
73
 
@@ -67,7 +84,15 @@ export function createRateLimiter () {
67
84
  getBucket(key).tokens += tokens
68
85
  }
69
86
 
70
- return { check, recordRequest, recordTokens }
87
+ /**
88
+ * @param {string} key
89
+ * @param {number} inputs
90
+ */
91
+ const recordInputs = (key, inputs) => {
92
+ getBucket(key).inputs += inputs
93
+ }
94
+
95
+ return { check, recordRequest, recordTokens, recordInputs }
71
96
  }
72
97
 
73
98
  // Single session-local instance.
@@ -75,3 +100,4 @@ const defaultLimiter = createRateLimiter()
75
100
  export const check = defaultLimiter.check
76
101
  export const recordRequest = defaultLimiter.recordRequest
77
102
  export const recordTokens = defaultLimiter.recordTokens
103
+ export const recordInputs = defaultLimiter.recordInputs
@@ -3,27 +3,46 @@
3
3
  * returns either a result or a typed error, never throwing.
4
4
  *
5
5
  * Mirrors `run_transcription.js`: one synchronous request, no streaming, no
6
- * cancellation path beyond the caller's own signal.
6
+ * cancellation path beyond the caller's own signal. Rate limits are enforced
7
+ * as in `run.js`, minus the speed lanes embeddings do not have.
7
8
  *
8
9
  * @module session/run_embedding
9
10
  */
10
11
 
11
12
  import { getEmbeddingAdapter } from './adapters/embedding/index.js'
12
13
  import { classifyProviderError } from './adapters/_errors.js'
14
+ import { getProviderLimits } from './adapters/_providers.js'
15
+ import * as defaultLimiter from './_rate_limiter.js'
13
16
  import { providerOf } from '#core/model-id.js'
14
17
 
15
18
  /**
16
19
  * @param {import('#core/embedding.js').EmbedEnvelope} envelope
17
- * @param {{resolveAdapter?: (provider: string) => any, spec?: any}} [options]
20
+ * @param {{
21
+ * resolveAdapter?: (provider: string) => any,
22
+ * resolveProviderLimits?: (provider: string) => any,
23
+ * limiter?: any,
24
+ * sleep?: (ms: number) => Promise<void>,
25
+ * modelKey?: string,
26
+ * spec?: any
27
+ * }} [options]
18
28
  * @returns {Promise<
19
29
  * | {ok: true, result: import('#core/embedding.js').EmbedResult}
20
30
  * | {ok: false, error: import('#core/errors.js').TypedError}
21
31
  * >}
22
32
  */
23
- export async function runEmbedding (envelope, { resolveAdapter = getEmbeddingAdapter, spec } = {}) {
33
+ export async function runEmbedding (envelope, {
34
+ resolveAdapter = getEmbeddingAdapter,
35
+ resolveProviderLimits = getProviderLimits,
36
+ limiter = defaultLimiter,
37
+ sleep = defaultSleep,
38
+ modelKey = envelope.model,
39
+ spec
40
+ } = {}) {
41
+ const provider = providerOf(envelope.model)
42
+
24
43
  let adapter
25
44
  try {
26
- adapter = resolveAdapter(providerOf(envelope.model))
45
+ adapter = resolveAdapter(provider)
27
46
  } catch (e) {
28
47
  return {
29
48
  ok: false,
@@ -36,8 +55,27 @@ export async function runEmbedding (envelope, { resolveAdapter = getEmbeddingAda
36
55
  }
37
56
  }
38
57
 
58
+ const providerCfg = resolveProviderLimits(provider) || {}
59
+ const rpmLimit = spec?.rpmLimit ?? providerCfg.rpmLimit
60
+ const tpmLimit = spec?.tpmLimit ?? providerCfg.tpmLimit
61
+ const inpmLimit = spec?.inpmLimit ?? providerCfg.inpmLimit
62
+ // `modelKey` is the catalog key, which the envelope carries over the wire but
63
+ // not on the in-process path, where it holds the upstream id instead.
64
+ const bucketKey = spec?.rateLimitScope === 'model' ? modelKey : provider
65
+ // A malformed envelope is metered as nothing: `checkBatch` rejects it inside
66
+ // the adapter, and a call that never reaches the provider must not spend quota.
67
+ const inputs = Array.isArray(envelope.input) ? envelope.input.length : 0
68
+
69
+ if (inputs > 0 && (rpmLimit != null || tpmLimit != null || inpmLimit != null)) {
70
+ const delay = limiter.check(bucketKey, { rpmLimit, tpmLimit, inpmLimit }, { inputs })
71
+ if (delay > 0) await sleep(delay)
72
+ limiter.recordRequest(bucketKey)
73
+ if (inpmLimit != null) limiter.recordInputs(bucketKey, inputs)
74
+ }
75
+
39
76
  try {
40
77
  const result = await adapter(envelope, spec ? { spec } : {})
78
+ if (tpmLimit != null && result.inputTokens) limiter.recordTokens(bucketKey, result.inputTokens)
41
79
  return { ok: true, result }
42
80
  } catch (e) {
43
81
  const typed = /** @type {any} */(e).typed || classifyProviderError(e, envelope.auth?.key)
@@ -49,3 +87,8 @@ export async function runEmbedding (envelope, { resolveAdapter = getEmbeddingAda
49
87
  function messageOf (e) {
50
88
  return e instanceof Error ? e.message : String(e)
51
89
  }
90
+
91
+ /** @param {number} ms */
92
+ function defaultSleep (ms) {
93
+ return new Promise(resolve => setTimeout(resolve, ms))
94
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mohdel",
3
- "version": "1.2.0",
3
+ "version": "1.3.1",
4
4
  "license": "MIT",
5
5
  "author": {
6
6
  "name": "Christophe Le Bars",
@@ -135,7 +135,7 @@
135
135
  "@opentelemetry/exporter-trace-otlp-grpc": "^0.222.0",
136
136
  "@opentelemetry/sdk-node": "^0.222.0",
137
137
  "chalk": "^6.0.0",
138
- "mohdel-thin-gate-linux-x64-gnu": "1.2.0"
138
+ "mohdel-thin-gate-linux-x64-gnu": "1.3.1"
139
139
  },
140
140
  "dependencies": {
141
141
  "@anthropic-ai/sdk": "^0.125.0",
@@ -38,6 +38,7 @@ const ARGUMENT = {
38
38
  'tag rm': ['model', 'tag'],
39
39
  'ratelimit show': ['model'],
40
40
  'ratelimit set': ['model'],
41
+ 'ratelimit provider set': ['provider'],
41
42
  'ratelimit rm': ['model']
42
43
  }
43
44
 
package/src/cli/index.js CHANGED
@@ -80,9 +80,9 @@ Commands:
80
80
  tag rm <model> <tag> Remove a tag
81
81
 
82
82
  ratelimit show <model|provider> Show effective limits (mo rl show)
83
- ratelimit set <model> [rpm] [tpm] Set model-level limits
83
+ ratelimit set <model> <limit> <value> Set limits: rpm, tpm, inpm
84
84
  ratelimit rm <model> Remove model-level limits
85
- ratelimit provider set <p> [rpm] [tpm] Set provider-level limits
85
+ ratelimit provider set <p> <limit> <v> Set provider-level limits
86
86
  ratelimit provider rm <p> Remove provider-level limits
87
87
 
88
88
  ask <provider/model> [prompt] One-shot inference (pipeable)
@@ -222,8 +222,16 @@ a silent billing error, not a crash. Accuracy matters more than completeness.
222
222
  field removed. Editing an existing model? Read its current entry out of the
223
223
  catalog file first and change only what you mean to. Step 2 below prints
224
224
  every removal, so check the diff before handing it over.
225
+ 7. **Some numbers describe the key, not the model.** Providers sort accounts
226
+ into standings — trial and production, numbered tiers, committed use — under
227
+ whatever name they give them, and publish a row per standing. Rate limits are
228
+ the obvious case; a price can be one too, and so can access to a model at
229
+ all. Two accounts can read the same published page and correctly write
230
+ different numbers, so for these a URL alone never settles a value. Establish
231
+ which standing this key has before writing one, and record it. If you cannot
232
+ establish it, leave the field out and say which one and why.
225
233
  ${local
226
- ? `7. **This installation has its own fields and tags** — see *Local conventions*
234
+ ? `8. **This installation has its own fields and tags** — see *Local conventions*
227
235
  below, and do not treat the field table as the whole story. A field marked
228
236
  *measured* has no page to read it off: run the command named for it. Never
229
237
  apply a tag whose required fields you cannot supply — leave the tag off and
@@ -297,6 +305,12 @@ from the docs page. Steps 1 and 3 only read, so run them as often as you need.
297
305
  Step 4 writes to the user's catalog and shows them the diff first — hand them
298
306
  the command, do not run it for them.
299
307
 
308
+ Anything that would change the catalog ends in something they can apply: a
309
+ candidate file and its \`mo model apply\`, or the exact \`mo\` command for the
310
+ change. A summary of what you found is not an outcome — if the ask was to
311
+ change something, the reply that contains no applyable artifact has not
312
+ answered it.
313
+
300
314
  ## Where to read the numbers
301
315
 
302
316
  ${names.map(referenceList).join('\n')}
@@ -320,6 +334,58 @@ someone who picked a provider *because* it was free should not be shown a
320
334
  table of dollar figures with no explanation. The rates apply once the free
321
335
  quota is gone.
322
336
 
337
+ ## Account-dependent numbers
338
+
339
+ Some published numbers describe the key rather than the model (hard rule 7).
340
+ Rate limits are the case you will meet most often, so the steps below are
341
+ written for them; a price taken from a row that varies by account standing
342
+ takes the same route. Start by reading the catalog — the models in it are the
343
+ only ones you need to look up, and an earlier pass may already have settled the
344
+ standing.
345
+
346
+ 1. **Establish the standing, once per provider.** Open the provider's limits
347
+ page and see how it divides accounts — most publish a column or a table per
348
+ account standing. Ask the user which one their key is on, using the words
349
+ that page uses for them; ask rather than picking the likely one and inviting
350
+ a correction. If a provider's limits do not vary by standing, there is
351
+ nothing to establish.
352
+ 2. **Read the row for that standing**, per model. Providers usually publish
353
+ limits per model, and a model on the page that this catalog does not carry
354
+ is not your problem.
355
+ 3. **Check the unit before you write.** mohdel counts three things: requests per
356
+ minute, tokens per minute, and inputs per minute for an embedding endpoint
357
+ metered in inputs. A limit published in any other unit — images, concurrent
358
+ jobs, a daily or monthly cap — does not convert, because the conversion
359
+ depends on how the caller batches and that is not yours to assume. Leave the
360
+ fields out for that model and tell the user the endpoint, the number and the
361
+ unit as published, so they can decide what to do.
362
+ 4. **Put each number at the level it is published at.** A limit that differs per
363
+ model goes in that model's entry, as \`rpmLimit\` / \`tpmLimit\` / \`inpmLimit\`
364
+ with \`rateLimitScope: "model"\`. The provider level is for one quota the whole
365
+ key shares across everything the provider sells — hand the user
366
+ \`mo rl provider set <provider> …\`, and set \`rateLimitScope: "provider"\` on the
367
+ entries drawing on it. A limit published **per endpoint** is neither: it binds
368
+ the models that call that endpoint and no others, so it goes on each of their
369
+ entries. At provider level it would claim the whole key is capped there,
370
+ which is false as soon as the provider sells anything else. A quota a speed
371
+ lane sells separately goes on that lane inside \`speeds\`, where it outranks
372
+ both.
373
+ 5. **Say which row you read, in your summary.** \`source\` and \`sourcedAt\` pin the
374
+ page and the date but not the row, and mohdel has no field for the standing —
375
+ so the record is what you tell the user. Name the standing and which models
376
+ took numbers from it. Do not improvise a home for it in the entry; if they
377
+ want it kept there, that is theirs to decide.
378
+ 6. **Report what you left out.** An unset limit is one that gets discovered as a
379
+ 429 in production. Say which models you skipped and why — no standing given,
380
+ unit that does not convert, nothing published.
381
+
382
+ Limits reach the catalog the same way prices do, through a candidate and
383
+ \`mo model apply\`. The exception is a provider-level quota, which is not a
384
+ catalog fact at all: \`mo rl provider set\` writes it to
385
+ \`~/.config/mohdel/providers.json\`, and like every writing command it is one you
386
+ hand over rather than run. \`mo rl show <model>\` prints what the runtime will
387
+ use once applied, and reads nothing but config.
388
+
323
389
  ## Entry kinds
324
390
 
325
391
  - **Text/vision model** — the default. \`inputFormat\` lists what it accepts.
@@ -4,24 +4,115 @@ import { parseJsonFlag, jsonOutputOne } from './json-output.js'
4
4
  // CLI logger: silent for noisy levels, console.error for errors and fatals.
5
5
  const cliLogger = { ...silent, error: console.error, fatal: console.error }
6
6
 
7
+ const LIMIT_NAMES = ['rpm', 'tpm', 'inpm']
8
+
9
+ /**
10
+ * `0` is a killswitch, not "unset", so read nullability rather than truth.
11
+ *
12
+ * @param {{rpmLimit?: number, tpmLimit?: number, inpmLimit?: number} | null | undefined} entry
13
+ * @returns {string[]}
14
+ */
15
+ function limitParts (entry) {
16
+ if (!entry) return []
17
+ return LIMIT_NAMES
18
+ .filter(name => entry[`${name}Limit`] != null)
19
+ .map(name => `${name}=${entry[`${name}Limit`]}`)
20
+ }
21
+
22
+ /**
23
+ * @param {string[]} cleared Limits named on the command line; empty means all.
24
+ * @param {string[]} parts What is left afterwards.
25
+ */
26
+ function clearedLine (cleared, parts) {
27
+ if (cleared.length === 0) return 'limits cleared'
28
+ return `${cleared.join(', ')} cleared; ${parts.length ? `${parts.join(' ')} remain` : 'no limits remain'}`
29
+ }
30
+
31
+ /** @param {string[]} names */
32
+ function parseLimitNames (names) {
33
+ for (const name of names) {
34
+ if (!LIMIT_NAMES.includes(name)) {
35
+ console.error(`Unknown limit '${name}'. Known: ${LIMIT_NAMES.join(', ')}`)
36
+ process.exit(1)
37
+ }
38
+ }
39
+ return names
40
+ }
41
+
42
+ /** @param {string} raw */
43
+ function toCount (raw) {
44
+ const n = parseInt(raw, 10)
45
+ if (!Number.isInteger(n) || n < 0 || String(n) !== String(raw).trim()) {
46
+ console.error(`'${raw}' is not a whole number`)
47
+ process.exit(1)
48
+ }
49
+ return n
50
+ }
51
+
52
+ /**
53
+ * Two forms. Named pairs — `rpm 60 inpm 2000` — reach every limit, including
54
+ * one on its own. The positional `<rpm> [tpm]` covers the common pair; a
55
+ * leading digit picks that form, since no limit is named one.
56
+ *
57
+ * @param {string[]} args
58
+ * @param {string} usage
59
+ * @returns {{rpm?: number, tpm?: number, inpm?: number}}
60
+ */
61
+ function parseLimits (args, usage) {
62
+ if (args.length === 0) { console.error(usage); process.exit(1) }
63
+
64
+ if (/^\d/.test(args[0])) {
65
+ const [rpm, tpm] = args
66
+ return tpm ? { rpm: toCount(rpm), tpm: toCount(tpm) } : { rpm: toCount(rpm) }
67
+ }
68
+
69
+ /** @type {Record<string, number>} */
70
+ const limits = {}
71
+ for (let i = 0; i < args.length; i += 2) {
72
+ const name = args[i]
73
+ if (!LIMIT_NAMES.includes(name)) {
74
+ console.error(`Unknown limit '${name}'. Known: ${LIMIT_NAMES.join(', ')}`)
75
+ process.exit(1)
76
+ }
77
+ if (args[i + 1] == null) { console.error(`'${name}' needs a value`); process.exit(1) }
78
+ limits[name] = toCount(args[i + 1])
79
+ }
80
+ return limits
81
+ }
82
+
7
83
  export async function runRateLimit (args) {
8
84
  const jsonFlag = parseJsonFlag(args)
9
- const [action, arg1, arg2, arg3] = args
85
+ const [action, arg1] = args
10
86
 
11
87
  if (!action || action === '-h' || action === '--help') {
12
88
  console.log(`mohdel ratelimit — manage rate limits
13
89
 
14
90
  Usage:
15
91
  ratelimit show <model|provider> [--json] Show effective limits
16
- ratelimit set <model> [rpm] [tpm] Set model-level limits
17
- ratelimit rm <model> Remove model-level limits
18
- ratelimit provider set <provider> [rpm] [tpm] Set provider-level limits
19
- ratelimit provider rm <provider> Remove provider-level limits
92
+ ratelimit set <model> <limit> <value> … Set limits by name
93
+ ratelimit set <model> <rpm> [tpm] Shortcut for the two common ones
94
+ ratelimit rm <model> [limit …] Remove limits, or all of them
95
+
96
+ A <model> may carry a speed lane — openai/gpt-x@fast — to reach the quota that
97
+ lane sells separately. A lane outranks the entry, and carries rpm and tpm only.
98
+ ratelimit provider set <provider> <limit> <value> …
99
+ ratelimit provider set <provider> <rpm> [tpm]
100
+ ratelimit provider rm <provider> [limit …] Remove limits, or all of them
101
+
102
+ Limits:
103
+ rpm requests per minute
104
+ tpm tokens per minute
105
+ inpm inputs per minute — what an embedding endpoint is metered in when
106
+ the provider counts inputs rather than requests or tokens
20
107
 
21
108
  Examples:
22
109
  ratelimit show anthropic Provider limits
23
110
  ratelimit show gemini/gemini-flash-latest Model limits, then provider
111
+ ratelimit set cohere/embed-v4.0 inpm 2000
112
+ ratelimit set gemini/gemini-flash-latest rpm 15 tpm 1000000
113
+ ratelimit set openai/gpt-x@fast rpm 200
24
114
  ratelimit set gemini/gemini-flash-latest 15 1000000
115
+ ratelimit rm cohere/embed-v4.0 inpm
25
116
  ratelimit provider set anthropic 60 100000
26
117
 
27
118
  Aliases:
@@ -49,35 +140,24 @@ Configuration:
49
140
  if (providerAction === 'show') {
50
141
  if (!providerName) { console.error('Usage: ratelimit provider show <provider>'); process.exit(1) }
51
142
  const entry = mo.getProviderRateLimit(providerName)
52
- if (!entry) {
53
- console.log(`${providerName}: no limits set`)
54
- } else {
55
- const parts = []
56
- if (entry.rpmLimit) parts.push(`rpm=${entry.rpmLimit}`)
57
- if (entry.tpmLimit) parts.push(`tpm=${entry.tpmLimit}`)
58
- console.log(`${providerName}: ${parts.join(' ')}`)
59
- }
143
+ const parts = limitParts(entry)
144
+ console.log(parts.length ? `${providerName}: ${parts.join(' ')}` : `${providerName}: no limits set`)
60
145
  return
61
146
  }
62
147
 
63
148
  if (providerAction === 'set') {
64
149
  if (!providerName) { console.error('Usage: ratelimit provider set <provider> [rpm] [tpm]'); process.exit(1) }
65
- const [rpmStr, tpmStr] = providerArgs
66
- const rpm = rpmStr ? parseInt(rpmStr, 10) : undefined
67
- const tpm = tpmStr ? parseInt(tpmStr, 10) : undefined
68
- if (rpm == null && tpm == null) { console.error('Provide at least rpm or tpm'); process.exit(1) }
69
- const result = await mo.setProviderRateLimit(providerName, { rpm, tpm })
70
- const parts = []
71
- if (result.rpmLimit) parts.push(`rpm=${result.rpmLimit}`)
72
- if (result.tpmLimit) parts.push(`tpm=${result.tpmLimit}`)
73
- console.log(`${providerName}: ${parts.join(' ')}`)
150
+ const limits = parseLimits(providerArgs, 'Usage: ratelimit provider set <provider> <limit> <value> … | <rpm> [tpm]')
151
+ const result = await mo.setProviderRateLimit(providerName, limits)
152
+ console.log(`${providerName}: ${limitParts(result).join(' ')}`)
74
153
  return
75
154
  }
76
155
 
77
156
  if (providerAction === 'rm' || providerAction === 'remove') {
78
- if (!providerName) { console.error('Usage: ratelimit provider rm <provider>'); process.exit(1) }
79
- await mo.clearProviderRateLimit(providerName)
80
- console.log(`${providerName}: limits cleared`)
157
+ if (!providerName) { console.error('Usage: ratelimit provider rm <provider> [limit …]'); process.exit(1) }
158
+ const names = parseLimitNames(providerArgs)
159
+ const remaining = await mo.clearProviderRateLimit(providerName, names)
160
+ console.log(`${providerName}: ${clearedLine(names, limitParts(remaining))}`)
81
161
  return
82
162
  }
83
163
 
@@ -96,29 +176,29 @@ Configuration:
96
176
  if (model) {
97
177
  const info = model.info()
98
178
  const providerEntry = mo.getProviderRateLimit(info.provider) || {}
99
- const rpmLimit = info.rpmLimit ?? providerEntry.rpmLimit
100
- const tpmLimit = info.tpmLimit ?? providerEntry.tpmLimit
101
- const scope = info.rateLimitScope || 'provider'
102
- const source = (info.rpmLimit || info.tpmLimit) ? 'model' : 'provider'
179
+ const lane = info.speed ? info.speeds?.[info.speed] ?? {} : {}
180
+ const rpmLimit = lane.rpmLimit ?? info.rpmLimit ?? providerEntry.rpmLimit
181
+ const tpmLimit = lane.tpmLimit ?? info.tpmLimit ?? providerEntry.tpmLimit
182
+ const inpmLimit = info.inpmLimit ?? providerEntry.inpmLimit
183
+ // A lane only gets its own bucket when it declares a limit; otherwise its
184
+ // traffic counts against the entry's, which is what `scope` then describes.
185
+ const scope = limitParts(lane).length ? `lane:${info.speed}` : (info.rateLimitScope || 'provider')
186
+ const source = limitParts(lane).length ? 'lane' : (limitParts(info).length ? 'model' : 'provider')
103
187
  if (jsonFlag.json) {
104
- jsonOutputOne({ id: arg1, rpmLimit: rpmLimit || null, tpmLimit: tpmLimit || null, scope, source })
188
+ jsonOutputOne({ id: arg1, rpmLimit: rpmLimit || null, tpmLimit: tpmLimit || null, inpmLimit: inpmLimit || null, scope, source })
105
189
  return
106
190
  }
107
- if (!rpmLimit && !tpmLimit) {
191
+ const parts = limitParts({ rpmLimit, tpmLimit, inpmLimit })
192
+ if (parts.length === 0) {
108
193
  console.log(`${arg1}: no limits`)
109
194
  } else {
110
- const parts = []
111
- if (rpmLimit) parts.push(`rpm=${rpmLimit}`)
112
- if (tpmLimit) parts.push(`tpm=${tpmLimit}`)
113
- parts.push(`scope=${scope}`)
114
- parts.push(`(${source})`)
115
- console.log(`${arg1}: ${parts.join(' ')}`)
195
+ console.log(`${arg1}: ${[...parts, `scope=${scope}`, `(${source})`].join(' ')}`)
116
196
  }
117
197
  } else {
118
198
  // Treat as provider name
119
199
  const entry = mo.getProviderRateLimit(arg1)
120
200
  if (jsonFlag.json) {
121
- jsonOutputOne({ provider: arg1, rpmLimit: entry?.rpmLimit || null, tpmLimit: entry?.tpmLimit || null })
201
+ jsonOutputOne({ provider: arg1, rpmLimit: entry?.rpmLimit || null, tpmLimit: entry?.tpmLimit || null, inpmLimit: entry?.inpmLimit || null })
122
202
  return
123
203
  }
124
204
  if (!entry) {
@@ -127,6 +207,7 @@ Configuration:
127
207
  const parts = []
128
208
  if (entry.rpmLimit) parts.push(`rpm=${entry.rpmLimit}`)
129
209
  if (entry.tpmLimit) parts.push(`tpm=${entry.tpmLimit}`)
210
+ if (entry.inpmLimit) parts.push(`inpm=${entry.inpmLimit}`)
130
211
  console.log(`${arg1}: ${parts.join(' ')}`)
131
212
  }
132
213
  }
@@ -134,24 +215,28 @@ Configuration:
134
215
  }
135
216
 
136
217
  if (action === 'set') {
137
- if (!arg1) { console.error('Usage: ratelimit set <model> [rpm] [tpm]'); process.exit(1) }
138
- const rpm = arg2 ? parseInt(arg2, 10) : undefined
139
- const tpm = arg3 ? parseInt(arg3, 10) : undefined
140
- if (rpm == null && tpm == null) { console.error('Provide at least rpm or tpm'); process.exit(1) }
218
+ const usage = 'Usage: ratelimit set <model> <limit> <value> … | <rpm> [tpm]'
219
+ if (!arg1) { console.error(usage); process.exit(1) }
220
+ const limits = parseLimits(args.slice(2), usage)
141
221
  const model = useModel(arg1)
142
- const result = await model.setRateLimit({ rpm, tpm })
143
- const parts = []
144
- if (result.rpmLimit) parts.push(`rpm=${result.rpmLimit}`)
145
- if (result.tpmLimit) parts.push(`tpm=${result.tpmLimit}`)
146
- console.log(`${arg1}: ${parts.join(' ')} scope=model`)
222
+ const lane = arg1.split('@')[1]
223
+ let result
224
+ try {
225
+ result = await model.setRateLimit(limits)
226
+ } catch (err) {
227
+ console.error(err.message)
228
+ process.exit(1)
229
+ }
230
+ console.log(`${arg1}: ${limitParts(result).join(' ')} scope=${lane ? `lane:${lane}` : 'model'}`)
147
231
  return
148
232
  }
149
233
 
150
234
  if (action === 'rm' || action === 'remove') {
151
- if (!arg1) { console.error('Usage: ratelimit rm <model>'); process.exit(1) }
235
+ if (!arg1) { console.error('Usage: ratelimit rm <model> [limit …]'); process.exit(1) }
236
+ const names = parseLimitNames(args.slice(2))
152
237
  const model = useModel(arg1)
153
- await model.clearRateLimit()
154
- console.log(`${arg1}: model limits cleared`)
238
+ const remaining = await model.clearRateLimit(names)
239
+ console.log(`${arg1}: ${clearedLine(names, limitParts(remaining))}`)
155
240
  return
156
241
  }
157
242
 
package/src/lib/index.js CHANGED
@@ -26,6 +26,8 @@ export const version = createRequire(import.meta.url)('../../package.json').vers
26
26
 
27
27
  const noop = () => {}
28
28
 
29
+ const LIMIT_FIELDS = ['rpmLimit', 'tpmLimit', 'inpmLimit']
30
+
29
31
  // Verbosity tiers — controls which mohdel internal log lines fire.
30
32
  //
31
33
  // 0 Anomaly-only. Failures, throttling, deprecation, server lifecycle.
@@ -413,30 +415,31 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
413
415
  return (providerName) => {
414
416
  const entry = providersConfig[providerName]
415
417
  if (!entry) return null
416
- const { rpmLimit, tpmLimit } = entry
417
- return (rpmLimit || tpmLimit) ? { rpmLimit, tpmLimit } : null
418
+ const { rpmLimit, tpmLimit, inpmLimit } = entry
419
+ return (rpmLimit || tpmLimit || inpmLimit) ? { rpmLimit, tpmLimit, inpmLimit } : null
418
420
  }
419
421
  }
420
422
 
421
423
  if (prop === 'setProviderRateLimit') {
422
- return async (providerName, { rpm, tpm } = {}) => {
424
+ return async (providerName, { rpm, tpm, inpm } = {}) => {
423
425
  const entry = providersConfig[providerName] || (providersConfig[providerName] = {})
424
426
  if (rpm != null) entry.rpmLimit = rpm
425
427
  if (tpm != null) entry.tpmLimit = tpm
428
+ if (inpm != null) entry.inpmLimit = inpm
426
429
  await saveProvidersConfig(providersConfig)
427
430
  return entry
428
431
  }
429
432
  }
430
433
 
431
434
  if (prop === 'clearProviderRateLimit') {
432
- return async (providerName) => {
435
+ return async (providerName, names = []) => {
433
436
  const entry = providersConfig[providerName]
434
- if (entry) {
435
- delete entry.rpmLimit
436
- delete entry.tpmLimit
437
- if (Object.keys(entry).length === 0) delete providersConfig[providerName]
438
- await saveProvidersConfig(providersConfig)
439
- }
437
+ if (!entry) return null
438
+ const fields = names.length ? names.map(n => `${n}Limit`) : LIMIT_FIELDS
439
+ for (const field of fields) delete entry[field]
440
+ if (Object.keys(entry).length === 0) delete providersConfig[providerName]
441
+ await saveProvidersConfig(providersConfig)
442
+ return providersConfig[providerName] || null
440
443
  }
441
444
  }
442
445
 
@@ -735,36 +738,63 @@ const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffor
735
738
  return runAnswerEmbedding({
736
739
  provider: modelSpec.provider,
737
740
  model: modelSpec.model ?? resolvedModelId.split('/').pop(),
741
+ modelKey: resolvedModelId,
738
742
  configuration,
739
743
  input,
740
744
  options,
741
745
  spec: modelSpec
742
- })
746
+ }, { limiter: rateLimiter, resolveProviderLimits })
743
747
  }
744
748
  }
745
749
 
746
750
  if (prop === 'setRateLimit') {
747
- return async ({ rpm, tpm } = {}) => {
751
+ return async ({ rpm, tpm, inpm } = {}) => {
748
752
  const curatedCache = getCuratedCacheSnapshot()
749
753
  const model = curatedCache[resolvedModelId] || (curatedCache[resolvedModelId] = { ...modelSpec })
754
+
755
+ if (aliasSpeed) {
756
+ if (inpm != null) {
757
+ throw new Error(
758
+ 'inpm is not a speed-lane limit — lanes carry rpmLimit and tpmLimit only. ' +
759
+ `Set it on the entry instead: mo rl set ${resolvedModelId} inpm <n>`
760
+ )
761
+ }
762
+ const lane = { ...model.speeds[aliasSpeed] }
763
+ if (rpm != null) lane.rpmLimit = rpm
764
+ if (tpm != null) lane.tpmLimit = tpm
765
+ model.speeds = { ...model.speeds, [aliasSpeed]: lane }
766
+ await persistCuratedCache()
767
+ return { rpmLimit: lane.rpmLimit, tpmLimit: lane.tpmLimit }
768
+ }
769
+
750
770
  if (rpm != null) model.rpmLimit = rpm
751
771
  if (tpm != null) model.tpmLimit = tpm
772
+ if (inpm != null) model.inpmLimit = inpm
752
773
  model.rateLimitScope = 'model'
753
774
  await persistCuratedCache()
754
- return { rpmLimit: model.rpmLimit, tpmLimit: model.tpmLimit }
775
+ return { rpmLimit: model.rpmLimit, tpmLimit: model.tpmLimit, inpmLimit: model.inpmLimit }
755
776
  }
756
777
  }
757
778
 
758
779
  if (prop === 'clearRateLimit') {
759
- return async () => {
780
+ return async (names = []) => {
760
781
  const curatedCache = getCuratedCacheSnapshot()
761
782
  const model = curatedCache[resolvedModelId]
762
- if (model) {
763
- delete model.rpmLimit
764
- delete model.tpmLimit
765
- delete model.rateLimitScope
783
+ if (!model) return {}
784
+ const fields = names.length ? names.map(n => `${n}Limit`) : LIMIT_FIELDS
785
+
786
+ if (aliasSpeed) {
787
+ const lane = { ...model.speeds[aliasSpeed] }
788
+ for (const field of fields) delete lane[field]
789
+ model.speeds = { ...model.speeds, [aliasSpeed]: lane }
766
790
  await persistCuratedCache()
791
+ return { rpmLimit: lane.rpmLimit, tpmLimit: lane.tpmLimit }
767
792
  }
793
+
794
+ for (const field of fields) delete model[field]
795
+ if (!LIMIT_FIELDS.some(field => model[field] != null)) delete model.rateLimitScope
796
+ await persistCuratedCache()
797
+ return { rpmLimit: model.rpmLimit, tpmLimit: model.tpmLimit, inpmLimit: model.inpmLimit }
768
798
  }
769
799
  }
770
800
 
@@ -811,7 +841,8 @@ const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffor
811
841
  if (prop === 'info') {
812
842
  return () => { // Sync
813
843
  const catalog = getCuratedCacheSnapshot()
814
- return catalog?.[resolvedModelId] ? { ...catalog[resolvedModelId] } : { ...modelSpec }
844
+ const entry = catalog?.[resolvedModelId] ? { ...catalog[resolvedModelId] } : { ...modelSpec }
845
+ return aliasSpeed ? { ...entry, speed: aliasSpeed } : entry
815
846
  }
816
847
  }
817
848
 
package/src/lib/schema.js CHANGED
@@ -69,6 +69,7 @@ const fieldDefs = {
69
69
  suspended: { type: 'string' },
70
70
  rpmLimit: { type: 'number' },
71
71
  tpmLimit: { type: 'number' },
72
+ inpmLimit: { type: 'number' },
72
73
  rateLimitScope: { type: 'string', validate: (v) => ['model', 'provider'].includes(v) ? null : 'must be "model" or "provider"' },
73
74
  outputCapStrategy: { type: 'string', validate: (v) => ['error', 'accept'].includes(v) ? null : "must be 'error' or 'accept'" },
74
75
  supportsTools: { type: 'boolean' },
@@ -1,50 +0,0 @@
1
- // Lightweight per-minute rate limiter.
2
- // Tracks RPM and TPM with minute-bucket granularity.
3
- // Throttles (delays) rather than rejects — returns ms to wait.
4
-
5
- const createRateLimiter = () => {
6
- // key → { count, tokens, minute }
7
- const buckets = new Map()
8
-
9
- const currentMinute = () => Math.floor(Date.now() / 60000)
10
-
11
- const getBucket = (key) => {
12
- const minute = currentMinute()
13
- const bucket = buckets.get(key)
14
- if (bucket && bucket.minute === minute) return bucket
15
- const fresh = { count: 0, tokens: 0, minute }
16
- buckets.set(key, fresh)
17
- return fresh
18
- }
19
-
20
- const msUntilNextMinute = (minute) => Math.max(0, (minute + 1) * 60000 - Date.now())
21
-
22
- // Returns ms to wait before sending (0 = go ahead)
23
- const check = (key, { rpmLimit, tpmLimit } = {}) => {
24
- if (!rpmLimit && !tpmLimit) return 0
25
- const bucket = getBucket(key)
26
- if (rpmLimit && bucket.count >= rpmLimit) {
27
- return msUntilNextMinute(bucket.minute)
28
- }
29
- if (tpmLimit && bucket.tokens >= tpmLimit) {
30
- return msUntilNextMinute(bucket.minute)
31
- }
32
- return 0
33
- }
34
-
35
- // Record a request count (call before sending — RPM tracking)
36
- const recordRequest = (key) => {
37
- const bucket = getBucket(key)
38
- bucket.count++
39
- }
40
-
41
- // Record token usage (call after response — TPM tracking)
42
- const recordTokens = (key, tokens) => {
43
- const bucket = getBucket(key)
44
- bucket.tokens += tokens
45
- }
46
-
47
- return { check, recordRequest, recordTokens }
48
- }
49
-
50
- export default createRateLimiter