mohdel 1.2.0 → 1.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/config/curated.schema.json +8 -3
- package/js/factory/bridge.js +9 -2
- package/js/session/_rate_limiter.js +33 -7
- package/js/session/run_embedding.js +47 -4
- package/package.json +2 -2
- package/src/cli/complete.js +1 -0
- package/src/cli/index.js +2 -2
- package/src/cli/instructions.js +67 -1
- package/src/cli/ratelimit.js +135 -50
- package/src/lib/index.js +50 -19
- package/src/lib/schema.js +1 -0
- package/src/lib/rate-limiter.js +0 -50
|
@@ -553,12 +553,17 @@
|
|
|
553
553
|
"rpmLimit": {
|
|
554
554
|
"type": "integer",
|
|
555
555
|
"minimum": 1,
|
|
556
|
-
"description": "Requests per minute. Overrides provider default."
|
|
556
|
+
"description": "Requests per minute this key may send. Overrides the provider-level default."
|
|
557
557
|
},
|
|
558
558
|
"tpmLimit": {
|
|
559
559
|
"type": "integer",
|
|
560
560
|
"minimum": 1,
|
|
561
|
-
"description": "Tokens per minute. Overrides provider default."
|
|
561
|
+
"description": "Tokens per minute this key may spend. Overrides the provider-level default."
|
|
562
|
+
},
|
|
563
|
+
"inpmLimit": {
|
|
564
|
+
"type": "integer",
|
|
565
|
+
"minimum": 1,
|
|
566
|
+
"description": "Inputs per minute this key may send to an embedding endpoint, for a provider that meters the endpoint in inputs rather than requests or tokens (Cohere publishes '2,000 inputs / min'). Counted exactly before dispatch from the batch size; a batch larger than the whole allowance is sent rather than delayed, since waiting cannot make it fit."
|
|
562
567
|
},
|
|
563
568
|
"rateLimitScope": {
|
|
564
569
|
"type": "string",
|
|
@@ -566,7 +571,7 @@
|
|
|
566
571
|
"model",
|
|
567
572
|
"provider"
|
|
568
573
|
],
|
|
569
|
-
"description": "'model' = private
|
|
574
|
+
"description": "Whose budget this key's calls draw on: 'model' = a private bucket for this entry, 'provider' = the pool shared with every other model of the provider."
|
|
570
575
|
},
|
|
571
576
|
"deprecated": {
|
|
572
577
|
"type": "string",
|
package/js/factory/bridge.js
CHANGED
|
@@ -187,6 +187,8 @@ export async function runAnswerTranscription ({ provider, model, configuration,
|
|
|
187
187
|
* @param {object} args
|
|
188
188
|
* @param {string} args.provider
|
|
189
189
|
* @param {string} args.model
|
|
190
|
+
* @param {string} [args.modelKey] Mohdel catalog key, for the rate-limit
|
|
191
|
+
* bucket when the entry is model-scoped.
|
|
190
192
|
* @param {any} args.configuration
|
|
191
193
|
* @param {string | string[]} args.input One text or a batch; normalized to an
|
|
192
194
|
* array so the result shape never
|
|
@@ -195,9 +197,10 @@ export async function runAnswerTranscription ({ provider, model, configuration,
|
|
|
195
197
|
* the envelope; `callId` / `authId` are
|
|
196
198
|
* transport metadata.
|
|
197
199
|
* @param {any} [args.spec]
|
|
200
|
+
* @param {BridgeDeps} [deps]
|
|
198
201
|
* @returns {Promise<any>}
|
|
199
202
|
*/
|
|
200
|
-
export async function runAnswerEmbedding ({ provider, model, configuration, input, options = {}, spec }) {
|
|
203
|
+
export async function runAnswerEmbedding ({ provider, model, modelKey, configuration, input, options = {}, spec }, deps = {}) {
|
|
201
204
|
const callId = options.callId || newCallId()
|
|
202
205
|
const authId = options.authId || 'local'
|
|
203
206
|
assertValidIds(callId, authId, `${provider}/${model}`)
|
|
@@ -212,7 +215,11 @@ export async function runAnswerEmbedding ({ provider, model, configuration, inpu
|
|
|
212
215
|
if (options.inputType) envelope.inputType = options.inputType
|
|
213
216
|
if (options.dimensions !== undefined) envelope.dimensions = options.dimensions
|
|
214
217
|
|
|
215
|
-
const out = await runEmbedding(envelope,
|
|
218
|
+
const out = await runEmbedding(envelope, {
|
|
219
|
+
...deps,
|
|
220
|
+
...(modelKey ? { modelKey } : {}),
|
|
221
|
+
...(spec ? { spec } : {})
|
|
222
|
+
})
|
|
216
223
|
if (!out.ok) throw MohdelError.fromJSON(out.error, { provider, model })
|
|
217
224
|
return out.result
|
|
218
225
|
}
|
|
@@ -1,7 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Minute-bucket rate limiter (per-key: provider or provider/model).
|
|
3
3
|
*
|
|
4
|
-
* Tracks RPM and
|
|
4
|
+
* Tracks RPM, TPM and INPM — inputs per minute, the unit an embedding
|
|
5
|
+
* endpoint is metered in when the provider counts inputs rather than
|
|
6
|
+
* requests or tokens. Returns ms to wait if over limit — throttles
|
|
5
7
|
* rather than rejecting, so the caller can absorb small bursts
|
|
6
8
|
* without a 429 round-trip.
|
|
7
9
|
*
|
|
@@ -14,7 +16,7 @@
|
|
|
14
16
|
*/
|
|
15
17
|
|
|
16
18
|
export function createRateLimiter () {
|
|
17
|
-
/** @type {Map<string, {count: number, tokens: number, minute: number}>} */
|
|
19
|
+
/** @type {Map<string, {count: number, tokens: number, inputs: number, minute: number}>} */
|
|
18
20
|
const buckets = new Map()
|
|
19
21
|
|
|
20
22
|
const currentMinute = () => Math.floor(Date.now() / 60000)
|
|
@@ -24,7 +26,7 @@ export function createRateLimiter () {
|
|
|
24
26
|
const minute = currentMinute()
|
|
25
27
|
const b = buckets.get(key)
|
|
26
28
|
if (b && b.minute === minute) return b
|
|
27
|
-
const fresh = { count: 0, tokens: 0, minute }
|
|
29
|
+
const fresh = { count: 0, tokens: 0, inputs: 0, minute }
|
|
28
30
|
buckets.set(key, fresh)
|
|
29
31
|
return fresh
|
|
30
32
|
}
|
|
@@ -42,15 +44,30 @@ export function createRateLimiter () {
|
|
|
42
44
|
* returned regardless of the current bucket.
|
|
43
45
|
* - positive number → throttle at that value.
|
|
44
46
|
*
|
|
47
|
+
* `rpmLimit` and `tpmLimit` gate on what the bucket already holds:
|
|
48
|
+
* the size of the call ahead is not known until it returns.
|
|
49
|
+
* `inpmLimit` gates on what the call is about to add, which
|
|
50
|
+
* `pending.inputs` carries — a batch size is exact before dispatch,
|
|
51
|
+
* so the call is admitted only if the whole batch fits.
|
|
52
|
+
*
|
|
45
53
|
* @param {string} key
|
|
46
|
-
* @param {{rpmLimit?: number, tpmLimit?: number}} limits
|
|
54
|
+
* @param {{rpmLimit?: number, tpmLimit?: number, inpmLimit?: number}} limits
|
|
55
|
+
* @param {{inputs?: number}} [pending]
|
|
47
56
|
* @returns {number}
|
|
48
57
|
*/
|
|
49
|
-
const check = (key, { rpmLimit, tpmLimit } = {}) => {
|
|
50
|
-
if (rpmLimit == null && tpmLimit == null) return 0
|
|
58
|
+
const check = (key, { rpmLimit, tpmLimit, inpmLimit } = {}, pending = {}) => {
|
|
59
|
+
if (rpmLimit == null && tpmLimit == null && inpmLimit == null) return 0
|
|
51
60
|
const b = getBucket(key)
|
|
52
61
|
if (rpmLimit != null && b.count >= rpmLimit) return msUntilNextMinute(b.minute)
|
|
53
62
|
if (tpmLimit != null && b.tokens >= tpmLimit) return msUntilNextMinute(b.minute)
|
|
63
|
+
if (inpmLimit != null) {
|
|
64
|
+
const adding = pending.inputs ?? 0
|
|
65
|
+
if (inpmLimit === 0) return msUntilNextMinute(b.minute)
|
|
66
|
+
// A batch larger than the whole allowance never fits, so waiting out the
|
|
67
|
+
// minute buys nothing: send it and take the provider's answer. Splitting
|
|
68
|
+
// it is the caller's call, not mohdel's.
|
|
69
|
+
if (adding <= inpmLimit && b.inputs + adding > inpmLimit) return msUntilNextMinute(b.minute)
|
|
70
|
+
}
|
|
54
71
|
return 0
|
|
55
72
|
}
|
|
56
73
|
|
|
@@ -67,7 +84,15 @@ export function createRateLimiter () {
|
|
|
67
84
|
getBucket(key).tokens += tokens
|
|
68
85
|
}
|
|
69
86
|
|
|
70
|
-
|
|
87
|
+
/**
|
|
88
|
+
* @param {string} key
|
|
89
|
+
* @param {number} inputs
|
|
90
|
+
*/
|
|
91
|
+
const recordInputs = (key, inputs) => {
|
|
92
|
+
getBucket(key).inputs += inputs
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
return { check, recordRequest, recordTokens, recordInputs }
|
|
71
96
|
}
|
|
72
97
|
|
|
73
98
|
// Single session-local instance.
|
|
@@ -75,3 +100,4 @@ const defaultLimiter = createRateLimiter()
|
|
|
75
100
|
export const check = defaultLimiter.check
|
|
76
101
|
export const recordRequest = defaultLimiter.recordRequest
|
|
77
102
|
export const recordTokens = defaultLimiter.recordTokens
|
|
103
|
+
export const recordInputs = defaultLimiter.recordInputs
|
|
@@ -3,27 +3,46 @@
|
|
|
3
3
|
* returns either a result or a typed error, never throwing.
|
|
4
4
|
*
|
|
5
5
|
* Mirrors `run_transcription.js`: one synchronous request, no streaming, no
|
|
6
|
-
* cancellation path beyond the caller's own signal.
|
|
6
|
+
* cancellation path beyond the caller's own signal. Rate limits are enforced
|
|
7
|
+
* as in `run.js`, minus the speed lanes embeddings do not have.
|
|
7
8
|
*
|
|
8
9
|
* @module session/run_embedding
|
|
9
10
|
*/
|
|
10
11
|
|
|
11
12
|
import { getEmbeddingAdapter } from './adapters/embedding/index.js'
|
|
12
13
|
import { classifyProviderError } from './adapters/_errors.js'
|
|
14
|
+
import { getProviderLimits } from './adapters/_providers.js'
|
|
15
|
+
import * as defaultLimiter from './_rate_limiter.js'
|
|
13
16
|
import { providerOf } from '#core/model-id.js'
|
|
14
17
|
|
|
15
18
|
/**
|
|
16
19
|
* @param {import('#core/embedding.js').EmbedEnvelope} envelope
|
|
17
|
-
* @param {{
|
|
20
|
+
* @param {{
|
|
21
|
+
* resolveAdapter?: (provider: string) => any,
|
|
22
|
+
* resolveProviderLimits?: (provider: string) => any,
|
|
23
|
+
* limiter?: any,
|
|
24
|
+
* sleep?: (ms: number) => Promise<void>,
|
|
25
|
+
* modelKey?: string,
|
|
26
|
+
* spec?: any
|
|
27
|
+
* }} [options]
|
|
18
28
|
* @returns {Promise<
|
|
19
29
|
* | {ok: true, result: import('#core/embedding.js').EmbedResult}
|
|
20
30
|
* | {ok: false, error: import('#core/errors.js').TypedError}
|
|
21
31
|
* >}
|
|
22
32
|
*/
|
|
23
|
-
export async function runEmbedding (envelope, {
|
|
33
|
+
export async function runEmbedding (envelope, {
|
|
34
|
+
resolveAdapter = getEmbeddingAdapter,
|
|
35
|
+
resolveProviderLimits = getProviderLimits,
|
|
36
|
+
limiter = defaultLimiter,
|
|
37
|
+
sleep = defaultSleep,
|
|
38
|
+
modelKey = envelope.model,
|
|
39
|
+
spec
|
|
40
|
+
} = {}) {
|
|
41
|
+
const provider = providerOf(envelope.model)
|
|
42
|
+
|
|
24
43
|
let adapter
|
|
25
44
|
try {
|
|
26
|
-
adapter = resolveAdapter(
|
|
45
|
+
adapter = resolveAdapter(provider)
|
|
27
46
|
} catch (e) {
|
|
28
47
|
return {
|
|
29
48
|
ok: false,
|
|
@@ -36,8 +55,27 @@ export async function runEmbedding (envelope, { resolveAdapter = getEmbeddingAda
|
|
|
36
55
|
}
|
|
37
56
|
}
|
|
38
57
|
|
|
58
|
+
const providerCfg = resolveProviderLimits(provider) || {}
|
|
59
|
+
const rpmLimit = spec?.rpmLimit ?? providerCfg.rpmLimit
|
|
60
|
+
const tpmLimit = spec?.tpmLimit ?? providerCfg.tpmLimit
|
|
61
|
+
const inpmLimit = spec?.inpmLimit ?? providerCfg.inpmLimit
|
|
62
|
+
// `modelKey` is the catalog key, which the envelope carries over the wire but
|
|
63
|
+
// not on the in-process path, where it holds the upstream id instead.
|
|
64
|
+
const bucketKey = spec?.rateLimitScope === 'model' ? modelKey : provider
|
|
65
|
+
// A malformed envelope is metered as nothing: `checkBatch` rejects it inside
|
|
66
|
+
// the adapter, and a call that never reaches the provider must not spend quota.
|
|
67
|
+
const inputs = Array.isArray(envelope.input) ? envelope.input.length : 0
|
|
68
|
+
|
|
69
|
+
if (inputs > 0 && (rpmLimit != null || tpmLimit != null || inpmLimit != null)) {
|
|
70
|
+
const delay = limiter.check(bucketKey, { rpmLimit, tpmLimit, inpmLimit }, { inputs })
|
|
71
|
+
if (delay > 0) await sleep(delay)
|
|
72
|
+
limiter.recordRequest(bucketKey)
|
|
73
|
+
if (inpmLimit != null) limiter.recordInputs(bucketKey, inputs)
|
|
74
|
+
}
|
|
75
|
+
|
|
39
76
|
try {
|
|
40
77
|
const result = await adapter(envelope, spec ? { spec } : {})
|
|
78
|
+
if (tpmLimit != null && result.inputTokens) limiter.recordTokens(bucketKey, result.inputTokens)
|
|
41
79
|
return { ok: true, result }
|
|
42
80
|
} catch (e) {
|
|
43
81
|
const typed = /** @type {any} */(e).typed || classifyProviderError(e, envelope.auth?.key)
|
|
@@ -49,3 +87,8 @@ export async function runEmbedding (envelope, { resolveAdapter = getEmbeddingAda
|
|
|
49
87
|
function messageOf (e) {
|
|
50
88
|
return e instanceof Error ? e.message : String(e)
|
|
51
89
|
}
|
|
90
|
+
|
|
91
|
+
/** @param {number} ms */
|
|
92
|
+
function defaultSleep (ms) {
|
|
93
|
+
return new Promise(resolve => setTimeout(resolve, ms))
|
|
94
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "mohdel",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.3.1",
|
|
4
4
|
"license": "MIT",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "Christophe Le Bars",
|
|
@@ -135,7 +135,7 @@
|
|
|
135
135
|
"@opentelemetry/exporter-trace-otlp-grpc": "^0.222.0",
|
|
136
136
|
"@opentelemetry/sdk-node": "^0.222.0",
|
|
137
137
|
"chalk": "^6.0.0",
|
|
138
|
-
"mohdel-thin-gate-linux-x64-gnu": "1.
|
|
138
|
+
"mohdel-thin-gate-linux-x64-gnu": "1.3.1"
|
|
139
139
|
},
|
|
140
140
|
"dependencies": {
|
|
141
141
|
"@anthropic-ai/sdk": "^0.125.0",
|
package/src/cli/complete.js
CHANGED
package/src/cli/index.js
CHANGED
|
@@ -80,9 +80,9 @@ Commands:
|
|
|
80
80
|
tag rm <model> <tag> Remove a tag
|
|
81
81
|
|
|
82
82
|
ratelimit show <model|provider> Show effective limits (mo rl show)
|
|
83
|
-
ratelimit set <model>
|
|
83
|
+
ratelimit set <model> <limit> <value> Set limits: rpm, tpm, inpm
|
|
84
84
|
ratelimit rm <model> Remove model-level limits
|
|
85
|
-
ratelimit provider set <p>
|
|
85
|
+
ratelimit provider set <p> <limit> <v> Set provider-level limits
|
|
86
86
|
ratelimit provider rm <p> Remove provider-level limits
|
|
87
87
|
|
|
88
88
|
ask <provider/model> [prompt] One-shot inference (pipeable)
|
package/src/cli/instructions.js
CHANGED
|
@@ -222,8 +222,16 @@ a silent billing error, not a crash. Accuracy matters more than completeness.
|
|
|
222
222
|
field removed. Editing an existing model? Read its current entry out of the
|
|
223
223
|
catalog file first and change only what you mean to. Step 2 below prints
|
|
224
224
|
every removal, so check the diff before handing it over.
|
|
225
|
+
7. **Some numbers describe the key, not the model.** Providers sort accounts
|
|
226
|
+
into standings — trial and production, numbered tiers, committed use — under
|
|
227
|
+
whatever name they give them, and publish a row per standing. Rate limits are
|
|
228
|
+
the obvious case; a price can be one too, and so can access to a model at
|
|
229
|
+
all. Two accounts can read the same published page and correctly write
|
|
230
|
+
different numbers, so for these a URL alone never settles a value. Establish
|
|
231
|
+
which standing this key has before writing one, and record it. If you cannot
|
|
232
|
+
establish it, leave the field out and say which one and why.
|
|
225
233
|
${local
|
|
226
|
-
? `
|
|
234
|
+
? `8. **This installation has its own fields and tags** — see *Local conventions*
|
|
227
235
|
below, and do not treat the field table as the whole story. A field marked
|
|
228
236
|
*measured* has no page to read it off: run the command named for it. Never
|
|
229
237
|
apply a tag whose required fields you cannot supply — leave the tag off and
|
|
@@ -297,6 +305,12 @@ from the docs page. Steps 1 and 3 only read, so run them as often as you need.
|
|
|
297
305
|
Step 4 writes to the user's catalog and shows them the diff first — hand them
|
|
298
306
|
the command, do not run it for them.
|
|
299
307
|
|
|
308
|
+
Anything that would change the catalog ends in something they can apply: a
|
|
309
|
+
candidate file and its \`mo model apply\`, or the exact \`mo\` command for the
|
|
310
|
+
change. A summary of what you found is not an outcome — if the ask was to
|
|
311
|
+
change something, the reply that contains no applyable artifact has not
|
|
312
|
+
answered it.
|
|
313
|
+
|
|
300
314
|
## Where to read the numbers
|
|
301
315
|
|
|
302
316
|
${names.map(referenceList).join('\n')}
|
|
@@ -320,6 +334,58 @@ someone who picked a provider *because* it was free should not be shown a
|
|
|
320
334
|
table of dollar figures with no explanation. The rates apply once the free
|
|
321
335
|
quota is gone.
|
|
322
336
|
|
|
337
|
+
## Account-dependent numbers
|
|
338
|
+
|
|
339
|
+
Some published numbers describe the key rather than the model (hard rule 7).
|
|
340
|
+
Rate limits are the case you will meet most often, so the steps below are
|
|
341
|
+
written for them; a price taken from a row that varies by account standing
|
|
342
|
+
takes the same route. Start by reading the catalog — the models in it are the
|
|
343
|
+
only ones you need to look up, and an earlier pass may already have settled the
|
|
344
|
+
standing.
|
|
345
|
+
|
|
346
|
+
1. **Establish the standing, once per provider.** Open the provider's limits
|
|
347
|
+
page and see how it divides accounts — most publish a column or a table per
|
|
348
|
+
account standing. Ask the user which one their key is on, using the words
|
|
349
|
+
that page uses for them; ask rather than picking the likely one and inviting
|
|
350
|
+
a correction. If a provider's limits do not vary by standing, there is
|
|
351
|
+
nothing to establish.
|
|
352
|
+
2. **Read the row for that standing**, per model. Providers usually publish
|
|
353
|
+
limits per model, and a model on the page that this catalog does not carry
|
|
354
|
+
is not your problem.
|
|
355
|
+
3. **Check the unit before you write.** mohdel counts three things: requests per
|
|
356
|
+
minute, tokens per minute, and inputs per minute for an embedding endpoint
|
|
357
|
+
metered in inputs. A limit published in any other unit — images, concurrent
|
|
358
|
+
jobs, a daily or monthly cap — does not convert, because the conversion
|
|
359
|
+
depends on how the caller batches and that is not yours to assume. Leave the
|
|
360
|
+
fields out for that model and tell the user the endpoint, the number and the
|
|
361
|
+
unit as published, so they can decide what to do.
|
|
362
|
+
4. **Put each number at the level it is published at.** A limit that differs per
|
|
363
|
+
model goes in that model's entry, as \`rpmLimit\` / \`tpmLimit\` / \`inpmLimit\`
|
|
364
|
+
with \`rateLimitScope: "model"\`. The provider level is for one quota the whole
|
|
365
|
+
key shares across everything the provider sells — hand the user
|
|
366
|
+
\`mo rl provider set <provider> …\`, and set \`rateLimitScope: "provider"\` on the
|
|
367
|
+
entries drawing on it. A limit published **per endpoint** is neither: it binds
|
|
368
|
+
the models that call that endpoint and no others, so it goes on each of their
|
|
369
|
+
entries. At provider level it would claim the whole key is capped there,
|
|
370
|
+
which is false as soon as the provider sells anything else. A quota a speed
|
|
371
|
+
lane sells separately goes on that lane inside \`speeds\`, where it outranks
|
|
372
|
+
both.
|
|
373
|
+
5. **Say which row you read, in your summary.** \`source\` and \`sourcedAt\` pin the
|
|
374
|
+
page and the date but not the row, and mohdel has no field for the standing —
|
|
375
|
+
so the record is what you tell the user. Name the standing and which models
|
|
376
|
+
took numbers from it. Do not improvise a home for it in the entry; if they
|
|
377
|
+
want it kept there, that is theirs to decide.
|
|
378
|
+
6. **Report what you left out.** An unset limit is one that gets discovered as a
|
|
379
|
+
429 in production. Say which models you skipped and why — no standing given,
|
|
380
|
+
unit that does not convert, nothing published.
|
|
381
|
+
|
|
382
|
+
Limits reach the catalog the same way prices do, through a candidate and
|
|
383
|
+
\`mo model apply\`. The exception is a provider-level quota, which is not a
|
|
384
|
+
catalog fact at all: \`mo rl provider set\` writes it to
|
|
385
|
+
\`~/.config/mohdel/providers.json\`, and like every writing command it is one you
|
|
386
|
+
hand over rather than run. \`mo rl show <model>\` prints what the runtime will
|
|
387
|
+
use once applied, and reads nothing but config.
|
|
388
|
+
|
|
323
389
|
## Entry kinds
|
|
324
390
|
|
|
325
391
|
- **Text/vision model** — the default. \`inputFormat\` lists what it accepts.
|
package/src/cli/ratelimit.js
CHANGED
|
@@ -4,24 +4,115 @@ import { parseJsonFlag, jsonOutputOne } from './json-output.js'
|
|
|
4
4
|
// CLI logger: silent for noisy levels, console.error for errors and fatals.
|
|
5
5
|
const cliLogger = { ...silent, error: console.error, fatal: console.error }
|
|
6
6
|
|
|
7
|
+
const LIMIT_NAMES = ['rpm', 'tpm', 'inpm']
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* `0` is a killswitch, not "unset", so read nullability rather than truth.
|
|
11
|
+
*
|
|
12
|
+
* @param {{rpmLimit?: number, tpmLimit?: number, inpmLimit?: number} | null | undefined} entry
|
|
13
|
+
* @returns {string[]}
|
|
14
|
+
*/
|
|
15
|
+
function limitParts (entry) {
|
|
16
|
+
if (!entry) return []
|
|
17
|
+
return LIMIT_NAMES
|
|
18
|
+
.filter(name => entry[`${name}Limit`] != null)
|
|
19
|
+
.map(name => `${name}=${entry[`${name}Limit`]}`)
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* @param {string[]} cleared Limits named on the command line; empty means all.
|
|
24
|
+
* @param {string[]} parts What is left afterwards.
|
|
25
|
+
*/
|
|
26
|
+
function clearedLine (cleared, parts) {
|
|
27
|
+
if (cleared.length === 0) return 'limits cleared'
|
|
28
|
+
return `${cleared.join(', ')} cleared; ${parts.length ? `${parts.join(' ')} remain` : 'no limits remain'}`
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/** @param {string[]} names */
|
|
32
|
+
function parseLimitNames (names) {
|
|
33
|
+
for (const name of names) {
|
|
34
|
+
if (!LIMIT_NAMES.includes(name)) {
|
|
35
|
+
console.error(`Unknown limit '${name}'. Known: ${LIMIT_NAMES.join(', ')}`)
|
|
36
|
+
process.exit(1)
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
return names
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
/** @param {string} raw */
|
|
43
|
+
function toCount (raw) {
|
|
44
|
+
const n = parseInt(raw, 10)
|
|
45
|
+
if (!Number.isInteger(n) || n < 0 || String(n) !== String(raw).trim()) {
|
|
46
|
+
console.error(`'${raw}' is not a whole number`)
|
|
47
|
+
process.exit(1)
|
|
48
|
+
}
|
|
49
|
+
return n
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Two forms. Named pairs — `rpm 60 inpm 2000` — reach every limit, including
|
|
54
|
+
* one on its own. The positional `<rpm> [tpm]` covers the common pair; a
|
|
55
|
+
* leading digit picks that form, since no limit is named one.
|
|
56
|
+
*
|
|
57
|
+
* @param {string[]} args
|
|
58
|
+
* @param {string} usage
|
|
59
|
+
* @returns {{rpm?: number, tpm?: number, inpm?: number}}
|
|
60
|
+
*/
|
|
61
|
+
function parseLimits (args, usage) {
|
|
62
|
+
if (args.length === 0) { console.error(usage); process.exit(1) }
|
|
63
|
+
|
|
64
|
+
if (/^\d/.test(args[0])) {
|
|
65
|
+
const [rpm, tpm] = args
|
|
66
|
+
return tpm ? { rpm: toCount(rpm), tpm: toCount(tpm) } : { rpm: toCount(rpm) }
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/** @type {Record<string, number>} */
|
|
70
|
+
const limits = {}
|
|
71
|
+
for (let i = 0; i < args.length; i += 2) {
|
|
72
|
+
const name = args[i]
|
|
73
|
+
if (!LIMIT_NAMES.includes(name)) {
|
|
74
|
+
console.error(`Unknown limit '${name}'. Known: ${LIMIT_NAMES.join(', ')}`)
|
|
75
|
+
process.exit(1)
|
|
76
|
+
}
|
|
77
|
+
if (args[i + 1] == null) { console.error(`'${name}' needs a value`); process.exit(1) }
|
|
78
|
+
limits[name] = toCount(args[i + 1])
|
|
79
|
+
}
|
|
80
|
+
return limits
|
|
81
|
+
}
|
|
82
|
+
|
|
7
83
|
export async function runRateLimit (args) {
|
|
8
84
|
const jsonFlag = parseJsonFlag(args)
|
|
9
|
-
const [action, arg1
|
|
85
|
+
const [action, arg1] = args
|
|
10
86
|
|
|
11
87
|
if (!action || action === '-h' || action === '--help') {
|
|
12
88
|
console.log(`mohdel ratelimit — manage rate limits
|
|
13
89
|
|
|
14
90
|
Usage:
|
|
15
91
|
ratelimit show <model|provider> [--json] Show effective limits
|
|
16
|
-
ratelimit set <model>
|
|
17
|
-
ratelimit
|
|
18
|
-
ratelimit
|
|
19
|
-
|
|
92
|
+
ratelimit set <model> <limit> <value> … Set limits by name
|
|
93
|
+
ratelimit set <model> <rpm> [tpm] Shortcut for the two common ones
|
|
94
|
+
ratelimit rm <model> [limit …] Remove limits, or all of them
|
|
95
|
+
|
|
96
|
+
A <model> may carry a speed lane — openai/gpt-x@fast — to reach the quota that
|
|
97
|
+
lane sells separately. A lane outranks the entry, and carries rpm and tpm only.
|
|
98
|
+
ratelimit provider set <provider> <limit> <value> …
|
|
99
|
+
ratelimit provider set <provider> <rpm> [tpm]
|
|
100
|
+
ratelimit provider rm <provider> [limit …] Remove limits, or all of them
|
|
101
|
+
|
|
102
|
+
Limits:
|
|
103
|
+
rpm requests per minute
|
|
104
|
+
tpm tokens per minute
|
|
105
|
+
inpm inputs per minute — what an embedding endpoint is metered in when
|
|
106
|
+
the provider counts inputs rather than requests or tokens
|
|
20
107
|
|
|
21
108
|
Examples:
|
|
22
109
|
ratelimit show anthropic Provider limits
|
|
23
110
|
ratelimit show gemini/gemini-flash-latest Model limits, then provider
|
|
111
|
+
ratelimit set cohere/embed-v4.0 inpm 2000
|
|
112
|
+
ratelimit set gemini/gemini-flash-latest rpm 15 tpm 1000000
|
|
113
|
+
ratelimit set openai/gpt-x@fast rpm 200
|
|
24
114
|
ratelimit set gemini/gemini-flash-latest 15 1000000
|
|
115
|
+
ratelimit rm cohere/embed-v4.0 inpm
|
|
25
116
|
ratelimit provider set anthropic 60 100000
|
|
26
117
|
|
|
27
118
|
Aliases:
|
|
@@ -49,35 +140,24 @@ Configuration:
|
|
|
49
140
|
if (providerAction === 'show') {
|
|
50
141
|
if (!providerName) { console.error('Usage: ratelimit provider show <provider>'); process.exit(1) }
|
|
51
142
|
const entry = mo.getProviderRateLimit(providerName)
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
} else {
|
|
55
|
-
const parts = []
|
|
56
|
-
if (entry.rpmLimit) parts.push(`rpm=${entry.rpmLimit}`)
|
|
57
|
-
if (entry.tpmLimit) parts.push(`tpm=${entry.tpmLimit}`)
|
|
58
|
-
console.log(`${providerName}: ${parts.join(' ')}`)
|
|
59
|
-
}
|
|
143
|
+
const parts = limitParts(entry)
|
|
144
|
+
console.log(parts.length ? `${providerName}: ${parts.join(' ')}` : `${providerName}: no limits set`)
|
|
60
145
|
return
|
|
61
146
|
}
|
|
62
147
|
|
|
63
148
|
if (providerAction === 'set') {
|
|
64
149
|
if (!providerName) { console.error('Usage: ratelimit provider set <provider> [rpm] [tpm]'); process.exit(1) }
|
|
65
|
-
const
|
|
66
|
-
const
|
|
67
|
-
|
|
68
|
-
if (rpm == null && tpm == null) { console.error('Provide at least rpm or tpm'); process.exit(1) }
|
|
69
|
-
const result = await mo.setProviderRateLimit(providerName, { rpm, tpm })
|
|
70
|
-
const parts = []
|
|
71
|
-
if (result.rpmLimit) parts.push(`rpm=${result.rpmLimit}`)
|
|
72
|
-
if (result.tpmLimit) parts.push(`tpm=${result.tpmLimit}`)
|
|
73
|
-
console.log(`${providerName}: ${parts.join(' ')}`)
|
|
150
|
+
const limits = parseLimits(providerArgs, 'Usage: ratelimit provider set <provider> <limit> <value> … | <rpm> [tpm]')
|
|
151
|
+
const result = await mo.setProviderRateLimit(providerName, limits)
|
|
152
|
+
console.log(`${providerName}: ${limitParts(result).join(' ')}`)
|
|
74
153
|
return
|
|
75
154
|
}
|
|
76
155
|
|
|
77
156
|
if (providerAction === 'rm' || providerAction === 'remove') {
|
|
78
|
-
if (!providerName) { console.error('Usage: ratelimit provider rm <provider>'); process.exit(1) }
|
|
79
|
-
|
|
80
|
-
|
|
157
|
+
if (!providerName) { console.error('Usage: ratelimit provider rm <provider> [limit …]'); process.exit(1) }
|
|
158
|
+
const names = parseLimitNames(providerArgs)
|
|
159
|
+
const remaining = await mo.clearProviderRateLimit(providerName, names)
|
|
160
|
+
console.log(`${providerName}: ${clearedLine(names, limitParts(remaining))}`)
|
|
81
161
|
return
|
|
82
162
|
}
|
|
83
163
|
|
|
@@ -96,29 +176,29 @@ Configuration:
|
|
|
96
176
|
if (model) {
|
|
97
177
|
const info = model.info()
|
|
98
178
|
const providerEntry = mo.getProviderRateLimit(info.provider) || {}
|
|
99
|
-
const
|
|
100
|
-
const
|
|
101
|
-
const
|
|
102
|
-
const
|
|
179
|
+
const lane = info.speed ? info.speeds?.[info.speed] ?? {} : {}
|
|
180
|
+
const rpmLimit = lane.rpmLimit ?? info.rpmLimit ?? providerEntry.rpmLimit
|
|
181
|
+
const tpmLimit = lane.tpmLimit ?? info.tpmLimit ?? providerEntry.tpmLimit
|
|
182
|
+
const inpmLimit = info.inpmLimit ?? providerEntry.inpmLimit
|
|
183
|
+
// A lane only gets its own bucket when it declares a limit; otherwise its
|
|
184
|
+
// traffic counts against the entry's, which is what `scope` then describes.
|
|
185
|
+
const scope = limitParts(lane).length ? `lane:${info.speed}` : (info.rateLimitScope || 'provider')
|
|
186
|
+
const source = limitParts(lane).length ? 'lane' : (limitParts(info).length ? 'model' : 'provider')
|
|
103
187
|
if (jsonFlag.json) {
|
|
104
|
-
jsonOutputOne({ id: arg1, rpmLimit: rpmLimit || null, tpmLimit: tpmLimit || null, scope, source })
|
|
188
|
+
jsonOutputOne({ id: arg1, rpmLimit: rpmLimit || null, tpmLimit: tpmLimit || null, inpmLimit: inpmLimit || null, scope, source })
|
|
105
189
|
return
|
|
106
190
|
}
|
|
107
|
-
|
|
191
|
+
const parts = limitParts({ rpmLimit, tpmLimit, inpmLimit })
|
|
192
|
+
if (parts.length === 0) {
|
|
108
193
|
console.log(`${arg1}: no limits`)
|
|
109
194
|
} else {
|
|
110
|
-
|
|
111
|
-
if (rpmLimit) parts.push(`rpm=${rpmLimit}`)
|
|
112
|
-
if (tpmLimit) parts.push(`tpm=${tpmLimit}`)
|
|
113
|
-
parts.push(`scope=${scope}`)
|
|
114
|
-
parts.push(`(${source})`)
|
|
115
|
-
console.log(`${arg1}: ${parts.join(' ')}`)
|
|
195
|
+
console.log(`${arg1}: ${[...parts, `scope=${scope}`, `(${source})`].join(' ')}`)
|
|
116
196
|
}
|
|
117
197
|
} else {
|
|
118
198
|
// Treat as provider name
|
|
119
199
|
const entry = mo.getProviderRateLimit(arg1)
|
|
120
200
|
if (jsonFlag.json) {
|
|
121
|
-
jsonOutputOne({ provider: arg1, rpmLimit: entry?.rpmLimit || null, tpmLimit: entry?.tpmLimit || null })
|
|
201
|
+
jsonOutputOne({ provider: arg1, rpmLimit: entry?.rpmLimit || null, tpmLimit: entry?.tpmLimit || null, inpmLimit: entry?.inpmLimit || null })
|
|
122
202
|
return
|
|
123
203
|
}
|
|
124
204
|
if (!entry) {
|
|
@@ -127,6 +207,7 @@ Configuration:
|
|
|
127
207
|
const parts = []
|
|
128
208
|
if (entry.rpmLimit) parts.push(`rpm=${entry.rpmLimit}`)
|
|
129
209
|
if (entry.tpmLimit) parts.push(`tpm=${entry.tpmLimit}`)
|
|
210
|
+
if (entry.inpmLimit) parts.push(`inpm=${entry.inpmLimit}`)
|
|
130
211
|
console.log(`${arg1}: ${parts.join(' ')}`)
|
|
131
212
|
}
|
|
132
213
|
}
|
|
@@ -134,24 +215,28 @@ Configuration:
|
|
|
134
215
|
}
|
|
135
216
|
|
|
136
217
|
if (action === 'set') {
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
const
|
|
140
|
-
if (rpm == null && tpm == null) { console.error('Provide at least rpm or tpm'); process.exit(1) }
|
|
218
|
+
const usage = 'Usage: ratelimit set <model> <limit> <value> … | <rpm> [tpm]'
|
|
219
|
+
if (!arg1) { console.error(usage); process.exit(1) }
|
|
220
|
+
const limits = parseLimits(args.slice(2), usage)
|
|
141
221
|
const model = useModel(arg1)
|
|
142
|
-
const
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
222
|
+
const lane = arg1.split('@')[1]
|
|
223
|
+
let result
|
|
224
|
+
try {
|
|
225
|
+
result = await model.setRateLimit(limits)
|
|
226
|
+
} catch (err) {
|
|
227
|
+
console.error(err.message)
|
|
228
|
+
process.exit(1)
|
|
229
|
+
}
|
|
230
|
+
console.log(`${arg1}: ${limitParts(result).join(' ')} scope=${lane ? `lane:${lane}` : 'model'}`)
|
|
147
231
|
return
|
|
148
232
|
}
|
|
149
233
|
|
|
150
234
|
if (action === 'rm' || action === 'remove') {
|
|
151
|
-
if (!arg1) { console.error('Usage: ratelimit rm <model>'); process.exit(1) }
|
|
235
|
+
if (!arg1) { console.error('Usage: ratelimit rm <model> [limit …]'); process.exit(1) }
|
|
236
|
+
const names = parseLimitNames(args.slice(2))
|
|
152
237
|
const model = useModel(arg1)
|
|
153
|
-
await model.clearRateLimit()
|
|
154
|
-
console.log(`${arg1}:
|
|
238
|
+
const remaining = await model.clearRateLimit(names)
|
|
239
|
+
console.log(`${arg1}: ${clearedLine(names, limitParts(remaining))}`)
|
|
155
240
|
return
|
|
156
241
|
}
|
|
157
242
|
|
package/src/lib/index.js
CHANGED
|
@@ -26,6 +26,8 @@ export const version = createRequire(import.meta.url)('../../package.json').vers
|
|
|
26
26
|
|
|
27
27
|
const noop = () => {}
|
|
28
28
|
|
|
29
|
+
const LIMIT_FIELDS = ['rpmLimit', 'tpmLimit', 'inpmLimit']
|
|
30
|
+
|
|
29
31
|
// Verbosity tiers — controls which mohdel internal log lines fire.
|
|
30
32
|
//
|
|
31
33
|
// 0 Anomaly-only. Failures, throttling, deprecation, server lifecycle.
|
|
@@ -413,30 +415,31 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
|
|
|
413
415
|
return (providerName) => {
|
|
414
416
|
const entry = providersConfig[providerName]
|
|
415
417
|
if (!entry) return null
|
|
416
|
-
const { rpmLimit, tpmLimit } = entry
|
|
417
|
-
return (rpmLimit || tpmLimit) ? { rpmLimit, tpmLimit } : null
|
|
418
|
+
const { rpmLimit, tpmLimit, inpmLimit } = entry
|
|
419
|
+
return (rpmLimit || tpmLimit || inpmLimit) ? { rpmLimit, tpmLimit, inpmLimit } : null
|
|
418
420
|
}
|
|
419
421
|
}
|
|
420
422
|
|
|
421
423
|
if (prop === 'setProviderRateLimit') {
|
|
422
|
-
return async (providerName, { rpm, tpm } = {}) => {
|
|
424
|
+
return async (providerName, { rpm, tpm, inpm } = {}) => {
|
|
423
425
|
const entry = providersConfig[providerName] || (providersConfig[providerName] = {})
|
|
424
426
|
if (rpm != null) entry.rpmLimit = rpm
|
|
425
427
|
if (tpm != null) entry.tpmLimit = tpm
|
|
428
|
+
if (inpm != null) entry.inpmLimit = inpm
|
|
426
429
|
await saveProvidersConfig(providersConfig)
|
|
427
430
|
return entry
|
|
428
431
|
}
|
|
429
432
|
}
|
|
430
433
|
|
|
431
434
|
if (prop === 'clearProviderRateLimit') {
|
|
432
|
-
return async (providerName) => {
|
|
435
|
+
return async (providerName, names = []) => {
|
|
433
436
|
const entry = providersConfig[providerName]
|
|
434
|
-
if (entry)
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
437
|
+
if (!entry) return null
|
|
438
|
+
const fields = names.length ? names.map(n => `${n}Limit`) : LIMIT_FIELDS
|
|
439
|
+
for (const field of fields) delete entry[field]
|
|
440
|
+
if (Object.keys(entry).length === 0) delete providersConfig[providerName]
|
|
441
|
+
await saveProvidersConfig(providersConfig)
|
|
442
|
+
return providersConfig[providerName] || null
|
|
440
443
|
}
|
|
441
444
|
}
|
|
442
445
|
|
|
@@ -735,36 +738,63 @@ const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffor
|
|
|
735
738
|
return runAnswerEmbedding({
|
|
736
739
|
provider: modelSpec.provider,
|
|
737
740
|
model: modelSpec.model ?? resolvedModelId.split('/').pop(),
|
|
741
|
+
modelKey: resolvedModelId,
|
|
738
742
|
configuration,
|
|
739
743
|
input,
|
|
740
744
|
options,
|
|
741
745
|
spec: modelSpec
|
|
742
|
-
})
|
|
746
|
+
}, { limiter: rateLimiter, resolveProviderLimits })
|
|
743
747
|
}
|
|
744
748
|
}
|
|
745
749
|
|
|
746
750
|
if (prop === 'setRateLimit') {
|
|
747
|
-
return async ({ rpm, tpm } = {}) => {
|
|
751
|
+
return async ({ rpm, tpm, inpm } = {}) => {
|
|
748
752
|
const curatedCache = getCuratedCacheSnapshot()
|
|
749
753
|
const model = curatedCache[resolvedModelId] || (curatedCache[resolvedModelId] = { ...modelSpec })
|
|
754
|
+
|
|
755
|
+
if (aliasSpeed) {
|
|
756
|
+
if (inpm != null) {
|
|
757
|
+
throw new Error(
|
|
758
|
+
'inpm is not a speed-lane limit — lanes carry rpmLimit and tpmLimit only. ' +
|
|
759
|
+
`Set it on the entry instead: mo rl set ${resolvedModelId} inpm <n>`
|
|
760
|
+
)
|
|
761
|
+
}
|
|
762
|
+
const lane = { ...model.speeds[aliasSpeed] }
|
|
763
|
+
if (rpm != null) lane.rpmLimit = rpm
|
|
764
|
+
if (tpm != null) lane.tpmLimit = tpm
|
|
765
|
+
model.speeds = { ...model.speeds, [aliasSpeed]: lane }
|
|
766
|
+
await persistCuratedCache()
|
|
767
|
+
return { rpmLimit: lane.rpmLimit, tpmLimit: lane.tpmLimit }
|
|
768
|
+
}
|
|
769
|
+
|
|
750
770
|
if (rpm != null) model.rpmLimit = rpm
|
|
751
771
|
if (tpm != null) model.tpmLimit = tpm
|
|
772
|
+
if (inpm != null) model.inpmLimit = inpm
|
|
752
773
|
model.rateLimitScope = 'model'
|
|
753
774
|
await persistCuratedCache()
|
|
754
|
-
return { rpmLimit: model.rpmLimit, tpmLimit: model.tpmLimit }
|
|
775
|
+
return { rpmLimit: model.rpmLimit, tpmLimit: model.tpmLimit, inpmLimit: model.inpmLimit }
|
|
755
776
|
}
|
|
756
777
|
}
|
|
757
778
|
|
|
758
779
|
if (prop === 'clearRateLimit') {
|
|
759
|
-
return async () => {
|
|
780
|
+
return async (names = []) => {
|
|
760
781
|
const curatedCache = getCuratedCacheSnapshot()
|
|
761
782
|
const model = curatedCache[resolvedModelId]
|
|
762
|
-
if (model) {
|
|
763
|
-
|
|
764
|
-
|
|
765
|
-
|
|
783
|
+
if (!model) return {}
|
|
784
|
+
const fields = names.length ? names.map(n => `${n}Limit`) : LIMIT_FIELDS
|
|
785
|
+
|
|
786
|
+
if (aliasSpeed) {
|
|
787
|
+
const lane = { ...model.speeds[aliasSpeed] }
|
|
788
|
+
for (const field of fields) delete lane[field]
|
|
789
|
+
model.speeds = { ...model.speeds, [aliasSpeed]: lane }
|
|
766
790
|
await persistCuratedCache()
|
|
791
|
+
return { rpmLimit: lane.rpmLimit, tpmLimit: lane.tpmLimit }
|
|
767
792
|
}
|
|
793
|
+
|
|
794
|
+
for (const field of fields) delete model[field]
|
|
795
|
+
if (!LIMIT_FIELDS.some(field => model[field] != null)) delete model.rateLimitScope
|
|
796
|
+
await persistCuratedCache()
|
|
797
|
+
return { rpmLimit: model.rpmLimit, tpmLimit: model.tpmLimit, inpmLimit: model.inpmLimit }
|
|
768
798
|
}
|
|
769
799
|
}
|
|
770
800
|
|
|
@@ -811,7 +841,8 @@ const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffor
|
|
|
811
841
|
if (prop === 'info') {
|
|
812
842
|
return () => { // Sync
|
|
813
843
|
const catalog = getCuratedCacheSnapshot()
|
|
814
|
-
|
|
844
|
+
const entry = catalog?.[resolvedModelId] ? { ...catalog[resolvedModelId] } : { ...modelSpec }
|
|
845
|
+
return aliasSpeed ? { ...entry, speed: aliasSpeed } : entry
|
|
815
846
|
}
|
|
816
847
|
}
|
|
817
848
|
|
package/src/lib/schema.js
CHANGED
|
@@ -69,6 +69,7 @@ const fieldDefs = {
|
|
|
69
69
|
suspended: { type: 'string' },
|
|
70
70
|
rpmLimit: { type: 'number' },
|
|
71
71
|
tpmLimit: { type: 'number' },
|
|
72
|
+
inpmLimit: { type: 'number' },
|
|
72
73
|
rateLimitScope: { type: 'string', validate: (v) => ['model', 'provider'].includes(v) ? null : 'must be "model" or "provider"' },
|
|
73
74
|
outputCapStrategy: { type: 'string', validate: (v) => ['error', 'accept'].includes(v) ? null : "must be 'error' or 'accept'" },
|
|
74
75
|
supportsTools: { type: 'boolean' },
|
package/src/lib/rate-limiter.js
DELETED
|
@@ -1,50 +0,0 @@
|
|
|
1
|
-
// Lightweight per-minute rate limiter.
|
|
2
|
-
// Tracks RPM and TPM with minute-bucket granularity.
|
|
3
|
-
// Throttles (delays) rather than rejects — returns ms to wait.
|
|
4
|
-
|
|
5
|
-
const createRateLimiter = () => {
|
|
6
|
-
// key → { count, tokens, minute }
|
|
7
|
-
const buckets = new Map()
|
|
8
|
-
|
|
9
|
-
const currentMinute = () => Math.floor(Date.now() / 60000)
|
|
10
|
-
|
|
11
|
-
const getBucket = (key) => {
|
|
12
|
-
const minute = currentMinute()
|
|
13
|
-
const bucket = buckets.get(key)
|
|
14
|
-
if (bucket && bucket.minute === minute) return bucket
|
|
15
|
-
const fresh = { count: 0, tokens: 0, minute }
|
|
16
|
-
buckets.set(key, fresh)
|
|
17
|
-
return fresh
|
|
18
|
-
}
|
|
19
|
-
|
|
20
|
-
const msUntilNextMinute = (minute) => Math.max(0, (minute + 1) * 60000 - Date.now())
|
|
21
|
-
|
|
22
|
-
// Returns ms to wait before sending (0 = go ahead)
|
|
23
|
-
const check = (key, { rpmLimit, tpmLimit } = {}) => {
|
|
24
|
-
if (!rpmLimit && !tpmLimit) return 0
|
|
25
|
-
const bucket = getBucket(key)
|
|
26
|
-
if (rpmLimit && bucket.count >= rpmLimit) {
|
|
27
|
-
return msUntilNextMinute(bucket.minute)
|
|
28
|
-
}
|
|
29
|
-
if (tpmLimit && bucket.tokens >= tpmLimit) {
|
|
30
|
-
return msUntilNextMinute(bucket.minute)
|
|
31
|
-
}
|
|
32
|
-
return 0
|
|
33
|
-
}
|
|
34
|
-
|
|
35
|
-
// Record a request count (call before sending — RPM tracking)
|
|
36
|
-
const recordRequest = (key) => {
|
|
37
|
-
const bucket = getBucket(key)
|
|
38
|
-
bucket.count++
|
|
39
|
-
}
|
|
40
|
-
|
|
41
|
-
// Record token usage (call after response — TPM tracking)
|
|
42
|
-
const recordTokens = (key, tokens) => {
|
|
43
|
-
const bucket = getBucket(key)
|
|
44
|
-
bucket.tokens += tokens
|
|
45
|
-
}
|
|
46
|
-
|
|
47
|
-
return { check, recordRequest, recordTokens }
|
|
48
|
-
}
|
|
49
|
-
|
|
50
|
-
export default createRateLimiter
|