mohdel 1.3.0 → 1.3.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +2 -2
- package/src/cli/instructions.js +67 -1
- package/src/cli/model.js +16 -2
- package/src/cli/ratelimit.js +20 -6
- package/src/lib/index.js +27 -1
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "mohdel",
|
|
3
|
-
"version": "1.3.
|
|
3
|
+
"version": "1.3.2",
|
|
4
4
|
"license": "MIT",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "Christophe Le Bars",
|
|
@@ -135,7 +135,7 @@
|
|
|
135
135
|
"@opentelemetry/exporter-trace-otlp-grpc": "^0.222.0",
|
|
136
136
|
"@opentelemetry/sdk-node": "^0.222.0",
|
|
137
137
|
"chalk": "^6.0.0",
|
|
138
|
-
"mohdel-thin-gate-linux-x64-gnu": "1.3.
|
|
138
|
+
"mohdel-thin-gate-linux-x64-gnu": "1.3.2"
|
|
139
139
|
},
|
|
140
140
|
"dependencies": {
|
|
141
141
|
"@anthropic-ai/sdk": "^0.125.0",
|
package/src/cli/instructions.js
CHANGED
|
@@ -222,8 +222,16 @@ a silent billing error, not a crash. Accuracy matters more than completeness.
|
|
|
222
222
|
field removed. Editing an existing model? Read its current entry out of the
|
|
223
223
|
catalog file first and change only what you mean to. Step 2 below prints
|
|
224
224
|
every removal, so check the diff before handing it over.
|
|
225
|
+
7. **Some numbers describe the key, not the model.** Providers sort accounts
|
|
226
|
+
into standings — trial and production, numbered tiers, committed use — under
|
|
227
|
+
whatever name they give them, and publish a row per standing. Rate limits are
|
|
228
|
+
the obvious case; a price can be one too, and so can access to a model at
|
|
229
|
+
all. Two accounts can read the same published page and correctly write
|
|
230
|
+
different numbers, so for these a URL alone never settles a value. Establish
|
|
231
|
+
which standing this key has before writing one, and record it. If you cannot
|
|
232
|
+
establish it, leave the field out and say which one and why.
|
|
225
233
|
${local
|
|
226
|
-
? `
|
|
234
|
+
? `8. **This installation has its own fields and tags** — see *Local conventions*
|
|
227
235
|
below, and do not treat the field table as the whole story. A field marked
|
|
228
236
|
*measured* has no page to read it off: run the command named for it. Never
|
|
229
237
|
apply a tag whose required fields you cannot supply — leave the tag off and
|
|
@@ -297,6 +305,12 @@ from the docs page. Steps 1 and 3 only read, so run them as often as you need.
|
|
|
297
305
|
Step 4 writes to the user's catalog and shows them the diff first — hand them
|
|
298
306
|
the command, do not run it for them.
|
|
299
307
|
|
|
308
|
+
Anything that would change the catalog ends in something they can apply: a
|
|
309
|
+
candidate file and its \`mo model apply\`, or the exact \`mo\` command for the
|
|
310
|
+
change. A summary of what you found is not an outcome — if the ask was to
|
|
311
|
+
change something, the reply that contains no applyable artifact has not
|
|
312
|
+
answered it.
|
|
313
|
+
|
|
300
314
|
## Where to read the numbers
|
|
301
315
|
|
|
302
316
|
${names.map(referenceList).join('\n')}
|
|
@@ -320,6 +334,58 @@ someone who picked a provider *because* it was free should not be shown a
|
|
|
320
334
|
table of dollar figures with no explanation. The rates apply once the free
|
|
321
335
|
quota is gone.
|
|
322
336
|
|
|
337
|
+
## Account-dependent numbers
|
|
338
|
+
|
|
339
|
+
Some published numbers describe the key rather than the model (hard rule 7).
|
|
340
|
+
Rate limits are the case you will meet most often, so the steps below are
|
|
341
|
+
written for them; a price taken from a row that varies by account standing
|
|
342
|
+
takes the same route. Start by reading the catalog — the models in it are the
|
|
343
|
+
only ones you need to look up, and an earlier pass may already have settled the
|
|
344
|
+
standing.
|
|
345
|
+
|
|
346
|
+
1. **Establish the standing, once per provider.** Open the provider's limits
|
|
347
|
+
page and see how it divides accounts — most publish a column or a table per
|
|
348
|
+
account standing. Ask the user which one their key is on, using the words
|
|
349
|
+
that page uses for them; ask rather than picking the likely one and inviting
|
|
350
|
+
a correction. If a provider's limits do not vary by standing, there is
|
|
351
|
+
nothing to establish.
|
|
352
|
+
2. **Read the row for that standing**, per model. Providers usually publish
|
|
353
|
+
limits per model, and a model on the page that this catalog does not carry
|
|
354
|
+
is not your problem.
|
|
355
|
+
3. **Check the unit before you write.** mohdel counts three things: requests per
|
|
356
|
+
minute, tokens per minute, and inputs per minute for an embedding endpoint
|
|
357
|
+
metered in inputs. A limit published in any other unit — images, concurrent
|
|
358
|
+
jobs, a daily or monthly cap — does not convert, because the conversion
|
|
359
|
+
depends on how the caller batches and that is not yours to assume. Leave the
|
|
360
|
+
fields out for that model and tell the user the endpoint, the number and the
|
|
361
|
+
unit as published, so they can decide what to do.
|
|
362
|
+
4. **Put each number at the level it is published at.** A limit that differs per
|
|
363
|
+
model goes in that model's entry, as \`rpmLimit\` / \`tpmLimit\` / \`inpmLimit\`
|
|
364
|
+
with \`rateLimitScope: "model"\`. The provider level is for one quota the whole
|
|
365
|
+
key shares across everything the provider sells — hand the user
|
|
366
|
+
\`mo rl provider set <provider> …\`, and set \`rateLimitScope: "provider"\` on the
|
|
367
|
+
entries drawing on it. A limit published **per endpoint** is neither: it binds
|
|
368
|
+
the models that call that endpoint and no others, so it goes on each of their
|
|
369
|
+
entries. At provider level it would claim the whole key is capped there,
|
|
370
|
+
which is false as soon as the provider sells anything else. A quota a speed
|
|
371
|
+
lane sells separately goes on that lane inside \`speeds\`, where it outranks
|
|
372
|
+
both.
|
|
373
|
+
5. **Say which row you read, in your summary.** \`source\` and \`sourcedAt\` pin the
|
|
374
|
+
page and the date but not the row, and mohdel has no field for the standing —
|
|
375
|
+
so the record is what you tell the user. Name the standing and which models
|
|
376
|
+
took numbers from it. Do not improvise a home for it in the entry; if they
|
|
377
|
+
want it kept there, that is theirs to decide.
|
|
378
|
+
6. **Report what you left out.** An unset limit is one that gets discovered as a
|
|
379
|
+
429 in production. Say which models you skipped and why — no standing given,
|
|
380
|
+
unit that does not convert, nothing published.
|
|
381
|
+
|
|
382
|
+
Limits reach the catalog the same way prices do, through a candidate and
|
|
383
|
+
\`mo model apply\`. The exception is a provider-level quota, which is not a
|
|
384
|
+
catalog fact at all: \`mo rl provider set\` writes it to
|
|
385
|
+
\`~/.config/mohdel/providers.json\`, and like every writing command it is one you
|
|
386
|
+
hand over rather than run. \`mo rl show <model>\` prints what the runtime will
|
|
387
|
+
use once applied, and reads nothing but config.
|
|
388
|
+
|
|
323
389
|
## Entry kinds
|
|
324
390
|
|
|
325
391
|
- **Text/vision model** — the default. \`inputFormat\` lists what it accepts.
|
package/src/cli/model.js
CHANGED
|
@@ -759,9 +759,23 @@ function resolvePrice (p) {
|
|
|
759
759
|
return 0
|
|
760
760
|
}
|
|
761
761
|
|
|
762
|
+
// Per-1M-token pricing is the pair; the other kinds bill on one dimension of
|
|
763
|
+
// their own, and reading only the pair prints a priced model as free.
|
|
764
|
+
const SINGLE_DIMENSION_PRICES = [
|
|
765
|
+
['embeddingPrice', ''],
|
|
766
|
+
['imagePrice', '/image'],
|
|
767
|
+
['transcriptionPrice', '/min']
|
|
768
|
+
]
|
|
769
|
+
|
|
762
770
|
function formatPrice (info) {
|
|
763
771
|
const inp = resolvePrice(info.inputPrice)
|
|
764
772
|
const out = resolvePrice(info.outputPrice)
|
|
765
|
-
if (
|
|
766
|
-
|
|
773
|
+
if (inp || out) return price(`$${inp}`) + meta('/') + price(`$${out}`)
|
|
774
|
+
|
|
775
|
+
for (const [field, unit] of SINGLE_DIMENSION_PRICES) {
|
|
776
|
+
const p = resolvePrice(info[field])
|
|
777
|
+
if (p) return price(`$${p}`) + meta(unit)
|
|
778
|
+
}
|
|
779
|
+
|
|
780
|
+
return meta('free')
|
|
767
781
|
}
|
package/src/cli/ratelimit.js
CHANGED
|
@@ -92,6 +92,9 @@ Usage:
|
|
|
92
92
|
ratelimit set <model> <limit> <value> … Set limits by name
|
|
93
93
|
ratelimit set <model> <rpm> [tpm] Shortcut for the two common ones
|
|
94
94
|
ratelimit rm <model> [limit …] Remove limits, or all of them
|
|
95
|
+
|
|
96
|
+
A <model> may carry a speed lane — openai/gpt-x@fast — to reach the quota that
|
|
97
|
+
lane sells separately. A lane outranks the entry, and carries rpm and tpm only.
|
|
95
98
|
ratelimit provider set <provider> <limit> <value> …
|
|
96
99
|
ratelimit provider set <provider> <rpm> [tpm]
|
|
97
100
|
ratelimit provider rm <provider> [limit …] Remove limits, or all of them
|
|
@@ -107,6 +110,7 @@ Examples:
|
|
|
107
110
|
ratelimit show gemini/gemini-flash-latest Model limits, then provider
|
|
108
111
|
ratelimit set cohere/embed-v4.0 inpm 2000
|
|
109
112
|
ratelimit set gemini/gemini-flash-latest rpm 15 tpm 1000000
|
|
113
|
+
ratelimit set openai/gpt-x@fast rpm 200
|
|
110
114
|
ratelimit set gemini/gemini-flash-latest 15 1000000
|
|
111
115
|
ratelimit rm cohere/embed-v4.0 inpm
|
|
112
116
|
ratelimit provider set anthropic 60 100000
|
|
@@ -172,11 +176,14 @@ Configuration:
|
|
|
172
176
|
if (model) {
|
|
173
177
|
const info = model.info()
|
|
174
178
|
const providerEntry = mo.getProviderRateLimit(info.provider) || {}
|
|
175
|
-
const
|
|
176
|
-
const
|
|
179
|
+
const lane = info.speed ? info.speeds?.[info.speed] ?? {} : {}
|
|
180
|
+
const rpmLimit = lane.rpmLimit ?? info.rpmLimit ?? providerEntry.rpmLimit
|
|
181
|
+
const tpmLimit = lane.tpmLimit ?? info.tpmLimit ?? providerEntry.tpmLimit
|
|
177
182
|
const inpmLimit = info.inpmLimit ?? providerEntry.inpmLimit
|
|
178
|
-
|
|
179
|
-
|
|
183
|
+
// A lane only gets its own bucket when it declares a limit; otherwise its
|
|
184
|
+
// traffic counts against the entry's, which is what `scope` then describes.
|
|
185
|
+
const scope = limitParts(lane).length ? `lane:${info.speed}` : (info.rateLimitScope || 'provider')
|
|
186
|
+
const source = limitParts(lane).length ? 'lane' : (limitParts(info).length ? 'model' : 'provider')
|
|
180
187
|
if (jsonFlag.json) {
|
|
181
188
|
jsonOutputOne({ id: arg1, rpmLimit: rpmLimit || null, tpmLimit: tpmLimit || null, inpmLimit: inpmLimit || null, scope, source })
|
|
182
189
|
return
|
|
@@ -212,8 +219,15 @@ Configuration:
|
|
|
212
219
|
if (!arg1) { console.error(usage); process.exit(1) }
|
|
213
220
|
const limits = parseLimits(args.slice(2), usage)
|
|
214
221
|
const model = useModel(arg1)
|
|
215
|
-
const
|
|
216
|
-
|
|
222
|
+
const lane = arg1.split('@')[1]
|
|
223
|
+
let result
|
|
224
|
+
try {
|
|
225
|
+
result = await model.setRateLimit(limits)
|
|
226
|
+
} catch (err) {
|
|
227
|
+
console.error(err.message)
|
|
228
|
+
process.exit(1)
|
|
229
|
+
}
|
|
230
|
+
console.log(`${arg1}: ${limitParts(result).join(' ')} scope=${lane ? `lane:${lane}` : 'model'}`)
|
|
217
231
|
return
|
|
218
232
|
}
|
|
219
233
|
|
package/src/lib/index.js
CHANGED
|
@@ -751,6 +751,22 @@ const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffor
|
|
|
751
751
|
return async ({ rpm, tpm, inpm } = {}) => {
|
|
752
752
|
const curatedCache = getCuratedCacheSnapshot()
|
|
753
753
|
const model = curatedCache[resolvedModelId] || (curatedCache[resolvedModelId] = { ...modelSpec })
|
|
754
|
+
|
|
755
|
+
if (aliasSpeed) {
|
|
756
|
+
if (inpm != null) {
|
|
757
|
+
throw new Error(
|
|
758
|
+
'inpm is not a speed-lane limit — lanes carry rpmLimit and tpmLimit only. ' +
|
|
759
|
+
`Set it on the entry instead: mo rl set ${resolvedModelId} inpm <n>`
|
|
760
|
+
)
|
|
761
|
+
}
|
|
762
|
+
const lane = { ...model.speeds[aliasSpeed] }
|
|
763
|
+
if (rpm != null) lane.rpmLimit = rpm
|
|
764
|
+
if (tpm != null) lane.tpmLimit = tpm
|
|
765
|
+
model.speeds = { ...model.speeds, [aliasSpeed]: lane }
|
|
766
|
+
await persistCuratedCache()
|
|
767
|
+
return { rpmLimit: lane.rpmLimit, tpmLimit: lane.tpmLimit }
|
|
768
|
+
}
|
|
769
|
+
|
|
754
770
|
if (rpm != null) model.rpmLimit = rpm
|
|
755
771
|
if (tpm != null) model.tpmLimit = tpm
|
|
756
772
|
if (inpm != null) model.inpmLimit = inpm
|
|
@@ -766,6 +782,15 @@ const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffor
|
|
|
766
782
|
const model = curatedCache[resolvedModelId]
|
|
767
783
|
if (!model) return {}
|
|
768
784
|
const fields = names.length ? names.map(n => `${n}Limit`) : LIMIT_FIELDS
|
|
785
|
+
|
|
786
|
+
if (aliasSpeed) {
|
|
787
|
+
const lane = { ...model.speeds[aliasSpeed] }
|
|
788
|
+
for (const field of fields) delete lane[field]
|
|
789
|
+
model.speeds = { ...model.speeds, [aliasSpeed]: lane }
|
|
790
|
+
await persistCuratedCache()
|
|
791
|
+
return { rpmLimit: lane.rpmLimit, tpmLimit: lane.tpmLimit }
|
|
792
|
+
}
|
|
793
|
+
|
|
769
794
|
for (const field of fields) delete model[field]
|
|
770
795
|
if (!LIMIT_FIELDS.some(field => model[field] != null)) delete model.rateLimitScope
|
|
771
796
|
await persistCuratedCache()
|
|
@@ -816,7 +841,8 @@ const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffor
|
|
|
816
841
|
if (prop === 'info') {
|
|
817
842
|
return () => { // Sync
|
|
818
843
|
const catalog = getCuratedCacheSnapshot()
|
|
819
|
-
|
|
844
|
+
const entry = catalog?.[resolvedModelId] ? { ...catalog[resolvedModelId] } : { ...modelSpec }
|
|
845
|
+
return aliasSpeed ? { ...entry, speed: aliasSpeed } : entry
|
|
820
846
|
}
|
|
821
847
|
}
|
|
822
848
|
|