mohdel 0.124.0 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +155 -30
- package/config/curated.schema.json +32 -10
- package/js/client/call.js +76 -19
- package/js/client/gate-binary.js +5 -0
- package/js/client/index.js +1 -0
- package/js/core/envelope.js +5 -1
- package/js/factory/bridge.js +2 -2
- package/js/session/adapters/_cancelled.js +0 -6
- package/js/session/adapters/_chat_completions.js +25 -0
- package/js/session/adapters/_output_cap.js +30 -0
- package/js/session/adapters/_registry.js +44 -0
- package/js/session/adapters/anthropic.js +8 -5
- package/js/session/adapters/gemini.js +4 -0
- package/js/session/adapters/openai.js +2 -1
- package/js/session/run.js +14 -4
- package/js/session/run_image.js +12 -3
- package/package.json +49 -19
- package/src/cli/aliases.js +20 -0
- package/src/cli/ask.js +58 -12
- package/src/cli/backup.js +2 -1
- package/src/cli/check.js +15 -86
- package/src/cli/complete.js +130 -0
- package/src/cli/default.js +34 -13
- package/src/cli/doctor.js +33 -14
- package/src/cli/entry.js +173 -0
- package/src/cli/index.js +77 -66
- package/src/cli/instructions.js +349 -0
- package/src/cli/local.js +14 -0
- package/src/cli/model.js +184 -37
- package/src/cli/onboard.js +186 -121
- package/src/cli/rank.js +2 -1
- package/src/cli/ratelimit.js +3 -3
- package/src/cli/tag.js +2 -0
- package/src/lib/assistants.js +93 -0
- package/src/lib/catalog/openrouter.js +6 -1
- package/src/lib/catalog-review.js +195 -0
- package/src/lib/common.js +14 -1
- package/src/lib/creators.js +35 -0
- package/src/lib/index.js +17 -3
- package/src/lib/local-conventions.js +120 -0
- package/src/lib/provider-info.js +98 -0
- package/src/lib/providers.js +69 -14
- package/src/lib/schema.js +15 -3
- package/src/lib/select.js +125 -67
- package/js/session/adapters/image/index.js +0 -40
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The output cap.
|
|
3
|
+
*
|
|
4
|
+
* `outputBudget` on the envelope is a request, not a promise. Providers
|
|
5
|
+
* disagree on what happens when it exceeds the model's own ceiling: some
|
|
6
|
+
* reject the call outright, others silently serve fewer tokens than asked
|
|
7
|
+
* for. Neither is useful to a caller, so mohdel never sends more than
|
|
8
|
+
* `spec.outputTokenLimit`.
|
|
9
|
+
*
|
|
10
|
+
* Adapters that add thinking headroom on top of the budget apply the cap
|
|
11
|
+
* *after* that addition — the sum is what reaches the wire, so the sum is
|
|
12
|
+
* what has to fit. Capping before would let the headroom push it back over.
|
|
13
|
+
*
|
|
14
|
+
* A spec carrying no `outputTokenLimit` cannot be capped and the caller's
|
|
15
|
+
* number is sent as given; that is the concrete cost of leaving the field
|
|
16
|
+
* out (docs/CATALOG.md).
|
|
17
|
+
*
|
|
18
|
+
* The catalog's `outputCapStrategy` records which of the two provider
|
|
19
|
+
* behaviours applies. It is informational — published for embedders that
|
|
20
|
+
* build their own provider requests rather than going through a session —
|
|
21
|
+
* and mohdel caps regardless of its value.
|
|
22
|
+
*
|
|
23
|
+
* @param {unknown} requested
|
|
24
|
+
* @param {unknown} outputTokenLimit
|
|
25
|
+
* @returns {unknown} `requested`, capped when both numbers are known.
|
|
26
|
+
*/
|
|
27
|
+
export const capOutput = (requested, outputTokenLimit) => {
|
|
28
|
+
if (typeof requested !== 'number' || typeof outputTokenLimit !== 'number') return requested
|
|
29
|
+
return Math.min(requested, outputTokenLimit)
|
|
30
|
+
}
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Adapter names and speed lanes, as plain data.
|
|
3
|
+
*
|
|
4
|
+
* Importing an adapter pulls its provider SDK; importing the registry in
|
|
5
|
+
* `index.js` pulls all of them, which is ~300ms. Anything that only needs to
|
|
6
|
+
* know *which* adapters exist, or what lanes a provider sells, reads this
|
|
7
|
+
* instead and pays nothing.
|
|
8
|
+
*
|
|
9
|
+
* @module session/adapters/registry
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
/** Every adapter module under `./`, named for the provider it serves. */
|
|
13
|
+
export const ADAPTER_NAMES = Object.freeze([
|
|
14
|
+
'anthropic',
|
|
15
|
+
'cerebras',
|
|
16
|
+
'deepseek',
|
|
17
|
+
'echo',
|
|
18
|
+
'fake',
|
|
19
|
+
'fireworks',
|
|
20
|
+
'gemini',
|
|
21
|
+
'groq',
|
|
22
|
+
'local',
|
|
23
|
+
'mistral',
|
|
24
|
+
'novita',
|
|
25
|
+
'openai',
|
|
26
|
+
'openrouter',
|
|
27
|
+
'qwen',
|
|
28
|
+
'xai',
|
|
29
|
+
'xiaomi'
|
|
30
|
+
])
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* Speed lanes each provider's adapter can emit. The adapter carries the same
|
|
34
|
+
* set on `adapter.speedLanes`; this is the copy readable without loading it.
|
|
35
|
+
*/
|
|
36
|
+
export const SPEED_LANES = Object.freeze({
|
|
37
|
+
openai: new Set(['fast', 'priority', 'flex', 'scale'])
|
|
38
|
+
})
|
|
39
|
+
|
|
40
|
+
/** Image adapters, keyed by provider; the module exports `<provider>Image`. */
|
|
41
|
+
export const IMAGE_ADAPTER_NAMES = Object.freeze(['openai', 'novita', 'fake'])
|
|
42
|
+
|
|
43
|
+
/** Whether a provider has an image adapter. Answerable without loading one. */
|
|
44
|
+
export const isImageProvider = (provider) => IMAGE_ADAPTER_NAMES.includes(provider)
|
|
@@ -15,6 +15,7 @@
|
|
|
15
15
|
*/
|
|
16
16
|
|
|
17
17
|
import Anthropic from '@anthropic-ai/sdk'
|
|
18
|
+
import { capOutput } from './_output_cap.js'
|
|
18
19
|
|
|
19
20
|
import {
|
|
20
21
|
STATUS_COMPLETED,
|
|
@@ -354,6 +355,8 @@ function buildRequest (envelope, conversation, system, conversationCacheTtl = nu
|
|
|
354
355
|
|
|
355
356
|
applyCacheBreakpoints(request, conversationCacheTtl)
|
|
356
357
|
|
|
358
|
+
request.max_tokens = capOutput(request.max_tokens, outputTokenLimit)
|
|
359
|
+
|
|
357
360
|
return request
|
|
358
361
|
}
|
|
359
362
|
|
|
@@ -453,11 +456,11 @@ function splitPrompt (prompt) {
|
|
|
453
456
|
}
|
|
454
457
|
}
|
|
455
458
|
if (m.role === 'system') {
|
|
456
|
-
// Translate
|
|
457
|
-
// Anthropic's cache_control. Preserves the block boundary
|
|
458
|
-
// chose; collapsing into a single string would silently disable
|
|
459
|
-
// caching even when the
|
|
460
|
-
//
|
|
459
|
+
// Translate caller-supplied cache markers ({text, cache: '5m'|'1h'})
|
|
460
|
+
// into Anthropic's cache_control. Preserves the block boundary the
|
|
461
|
+
// caller chose; collapsing into a single string would silently disable
|
|
462
|
+
// caching even when the prompt was composed with explicit
|
|
463
|
+
// breakpoints.
|
|
461
464
|
if (Array.isArray(m.content)) {
|
|
462
465
|
for (const p of m.content) {
|
|
463
466
|
if (!p?.text) continue
|
|
@@ -16,6 +16,7 @@
|
|
|
16
16
|
*/
|
|
17
17
|
|
|
18
18
|
import { GoogleGenAI } from '@google/genai'
|
|
19
|
+
import { capOutput } from './_output_cap.js'
|
|
19
20
|
|
|
20
21
|
import {
|
|
21
22
|
STATUS_COMPLETED,
|
|
@@ -265,6 +266,9 @@ function buildRequest (envelope, contents, systemInstruction) {
|
|
|
265
266
|
}
|
|
266
267
|
}
|
|
267
268
|
|
|
269
|
+
config.maxOutputTokens = capOutput(config.maxOutputTokens, spec?.outputTokenLimit)
|
|
270
|
+
if (config.maxOutputTokens === undefined) delete config.maxOutputTokens
|
|
271
|
+
|
|
268
272
|
/** @type {Record<string, any>} */
|
|
269
273
|
const request = {
|
|
270
274
|
model: spec?.model ?? bareOf(envelope.model),
|
|
@@ -16,6 +16,7 @@
|
|
|
16
16
|
*/
|
|
17
17
|
|
|
18
18
|
import OpenAI from 'openai'
|
|
19
|
+
import { SPEED_LANES } from './_registry.js'
|
|
19
20
|
|
|
20
21
|
import {
|
|
21
22
|
STATUS_COMPLETED,
|
|
@@ -230,7 +231,7 @@ export async function * openai (envelope, deps = {}) {
|
|
|
230
231
|
* Lane names the OpenAI adapter can put on `service_tier`. `run.js`
|
|
231
232
|
* reads this before dispatch; a lane outside it never reaches here.
|
|
232
233
|
*/
|
|
233
|
-
openai.speedLanes =
|
|
234
|
+
openai.speedLanes = SPEED_LANES.openai
|
|
234
235
|
|
|
235
236
|
/**
|
|
236
237
|
* Lane the response says was served, given the one requested.
|
package/js/session/run.js
CHANGED
|
@@ -22,8 +22,7 @@
|
|
|
22
22
|
* @module session/run
|
|
23
23
|
*/
|
|
24
24
|
|
|
25
|
-
import {
|
|
26
|
-
import { isImageProvider } from './adapters/image/index.js'
|
|
25
|
+
import { ADAPTER_NAMES, isImageProvider } from './adapters/_registry.js'
|
|
27
26
|
import { getSpec } from './adapters/_catalog.js'
|
|
28
27
|
import { getProviderLimits } from './adapters/_providers.js'
|
|
29
28
|
import { hasSpeed, mergeSpeed, speedHasOwnQuota, speedNames } from './adapters/_speed.js'
|
|
@@ -57,8 +56,19 @@ import { STATUS_INCOMPLETE, WARNING_CANCELLED } from '#core/status.js'
|
|
|
57
56
|
* }} [options]
|
|
58
57
|
* @returns {AsyncGenerator<import('#core/events.js').Event>}
|
|
59
58
|
*/
|
|
59
|
+
// Load one adapter rather than the registry: importing every adapter pulls
|
|
60
|
+
// every provider SDK, which is ~300ms a caller pays to use one of them. The
|
|
61
|
+
// name is checked against the known list before it reaches the import path —
|
|
62
|
+
// it comes off the envelope, and a computed specifier must never take an
|
|
63
|
+
// arbitrary string.
|
|
64
|
+
const loadAdapter = async (provider) => {
|
|
65
|
+
if (!ADAPTER_NAMES.includes(provider)) throw new Error(`unknown provider: ${provider}`)
|
|
66
|
+
const module = await import(`./adapters/${provider}.js`)
|
|
67
|
+
return module[provider]
|
|
68
|
+
}
|
|
69
|
+
|
|
60
70
|
export async function * run (envelope, {
|
|
61
|
-
resolveAdapter =
|
|
71
|
+
resolveAdapter = loadAdapter,
|
|
62
72
|
resolveSpec = getSpec,
|
|
63
73
|
resolveProviderLimits = getProviderLimits,
|
|
64
74
|
cooldown = defaultCooldown,
|
|
@@ -92,7 +102,7 @@ export async function * run (envelope, {
|
|
|
92
102
|
|
|
93
103
|
let adapter
|
|
94
104
|
try {
|
|
95
|
-
adapter = resolveAdapter(provider)
|
|
105
|
+
adapter = await resolveAdapter(provider)
|
|
96
106
|
} catch (e) {
|
|
97
107
|
// Distinguish "image-only provider invoked via answer" from
|
|
98
108
|
// truly-unknown. Novita-and-friends have no text adapter but a
|
package/js/session/run_image.js
CHANGED
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
* @module session/run_image
|
|
14
14
|
*/
|
|
15
15
|
|
|
16
|
-
import {
|
|
16
|
+
import { IMAGE_ADAPTER_NAMES } from './adapters/_registry.js'
|
|
17
17
|
import { classifyProviderError } from './adapters/_errors.js'
|
|
18
18
|
import { providerOf } from '#core/model-id.js'
|
|
19
19
|
|
|
@@ -31,10 +31,19 @@ import { providerOf } from '#core/model-id.js'
|
|
|
31
31
|
* | {ok: false, error: import('#core/errors.js').TypedError}
|
|
32
32
|
* >}
|
|
33
33
|
*/
|
|
34
|
-
|
|
34
|
+
// Loaded per provider rather than as a registry: the image adapters pull the
|
|
35
|
+
// OpenAI SDK, which a text-only caller never needs. Checked against the known
|
|
36
|
+
// list first — the name comes off the envelope.
|
|
37
|
+
const loadImageAdapter = async (provider) => {
|
|
38
|
+
if (!IMAGE_ADAPTER_NAMES.includes(provider)) throw new Error(`no image adapter for provider: ${provider}`)
|
|
39
|
+
const module = await import(`./adapters/image/${provider}.js`)
|
|
40
|
+
return module[`${provider}Image`]
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
export async function runImage (envelope, { resolveAdapter = loadImageAdapter, spec } = {}) {
|
|
35
44
|
let adapter
|
|
36
45
|
try {
|
|
37
|
-
adapter = resolveAdapter(providerOf(envelope.model))
|
|
46
|
+
adapter = await resolveAdapter(providerOf(envelope.model))
|
|
38
47
|
} catch (e) {
|
|
39
48
|
return {
|
|
40
49
|
ok: false,
|
package/package.json
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "mohdel",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "1.0.0",
|
|
4
4
|
"license": "MIT",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "Christophe Le Bars",
|
|
7
7
|
"email": "clb@toort.net"
|
|
8
8
|
},
|
|
9
|
-
"description": "Self-hosted LLM gateway and SDK for Node — a LiteLLM-style unified API for
|
|
9
|
+
"description": "Self-hosted LLM gateway and SDK for Node — a LiteLLM-style unified API for 13 providers (Anthropic, OpenAI, Gemini, Mistral, Groq, xAI, DeepSeek, OpenRouter, …) with per-call USD cost tracking, streaming, tool calls, vision, speech-to-text, and built-in OpenTelemetry. Run in-process, or behind the process-isolated thin-gate for fault containment.",
|
|
10
10
|
"type": "module",
|
|
11
11
|
"keywords": [
|
|
12
12
|
"llm",
|
|
@@ -42,14 +42,38 @@
|
|
|
42
42
|
},
|
|
43
43
|
"main": "src/lib/index.js",
|
|
44
44
|
"exports": {
|
|
45
|
-
".":
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
"./
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
45
|
+
".": {
|
|
46
|
+
"types": "./src/lib/index.d.ts",
|
|
47
|
+
"default": "./src/lib/index.js"
|
|
48
|
+
},
|
|
49
|
+
"./providers": {
|
|
50
|
+
"types": "./src/lib/providers.d.ts",
|
|
51
|
+
"default": "./src/lib/providers.js"
|
|
52
|
+
},
|
|
53
|
+
"./creators": {
|
|
54
|
+
"types": "./src/lib/creators.d.ts",
|
|
55
|
+
"default": "./src/lib/creators.js"
|
|
56
|
+
},
|
|
57
|
+
"./utils": {
|
|
58
|
+
"types": "./src/lib/utils.d.ts",
|
|
59
|
+
"default": "./src/lib/utils.js"
|
|
60
|
+
},
|
|
61
|
+
"./errors": {
|
|
62
|
+
"types": "./js/core/errors.d.ts",
|
|
63
|
+
"default": "./js/core/errors.js"
|
|
64
|
+
},
|
|
65
|
+
"./client": {
|
|
66
|
+
"types": "./js/client/index.d.ts",
|
|
67
|
+
"default": "./js/client/index.js"
|
|
68
|
+
},
|
|
69
|
+
"./session": {
|
|
70
|
+
"types": "./js/session/index.d.ts",
|
|
71
|
+
"default": "./js/session/index.js"
|
|
72
|
+
},
|
|
73
|
+
"./session/bin": {
|
|
74
|
+
"types": "./js/session/bin.d.ts",
|
|
75
|
+
"default": "./js/session/bin.js"
|
|
76
|
+
}
|
|
53
77
|
},
|
|
54
78
|
"imports": {
|
|
55
79
|
"#core": "./js/core/index.js",
|
|
@@ -64,7 +88,9 @@
|
|
|
64
88
|
"src/cli",
|
|
65
89
|
"config",
|
|
66
90
|
"README.md",
|
|
67
|
-
"LICENSE"
|
|
91
|
+
"LICENSE",
|
|
92
|
+
"js/**/*.d.ts",
|
|
93
|
+
"src/lib/**/*.d.ts"
|
|
68
94
|
],
|
|
69
95
|
"publishConfig": {
|
|
70
96
|
"registry": "https://registry.npmjs.org",
|
|
@@ -73,10 +99,12 @@
|
|
|
73
99
|
},
|
|
74
100
|
"scripts": {
|
|
75
101
|
"lint": "standard",
|
|
102
|
+
"clean:types": "find js src/lib -name '*.d.ts' -delete",
|
|
103
|
+
"build:types": "npm run clean:types && tsc -p tsconfig.build.json",
|
|
76
104
|
"test": "npm run test:js && npm run test:rust",
|
|
77
105
|
"test:js": "vitest run test/unit",
|
|
78
106
|
"test:rust": "cargo test --manifest-path rust/thin-gate/Cargo.toml && cargo test --manifest-path rust/napi-addon/Cargo.toml",
|
|
79
|
-
"prerelease": "npm run lint && npm run test",
|
|
107
|
+
"prerelease": "npm run lint && npm run build:types && npm run test",
|
|
80
108
|
"release": "release-it",
|
|
81
109
|
"test:provider": "vitest run test/integration/provider.test.js",
|
|
82
110
|
"test:multiturn": "vitest run test/integration/multiturn.test.js",
|
|
@@ -104,20 +132,20 @@
|
|
|
104
132
|
}
|
|
105
133
|
},
|
|
106
134
|
"optionalDependencies": {
|
|
107
|
-
"@clack/prompts": "^1.
|
|
135
|
+
"@clack/prompts": "^1.8.8",
|
|
108
136
|
"@opentelemetry/exporter-trace-otlp-grpc": "^0.222.0",
|
|
109
137
|
"@opentelemetry/sdk-node": "^0.222.0",
|
|
110
138
|
"chalk": "^6.0.0",
|
|
111
|
-
"mohdel-thin-gate-linux-x64-gnu": "0.
|
|
139
|
+
"mohdel-thin-gate-linux-x64-gnu": "1.0.0"
|
|
112
140
|
},
|
|
113
141
|
"dependencies": {
|
|
114
|
-
"@anthropic-ai/sdk": "^0.
|
|
142
|
+
"@anthropic-ai/sdk": "^0.125.0",
|
|
115
143
|
"@cerebras/cerebras_cloud_sdk": "^1.91.0",
|
|
116
|
-
"@google/genai": "^2.
|
|
144
|
+
"@google/genai": "^2.22.0",
|
|
117
145
|
"@opentelemetry/api": "^1.9.1",
|
|
118
146
|
"env-paths": "^4.0.0",
|
|
119
147
|
"groq-sdk": "^1.6.0",
|
|
120
|
-
"openai": "^7.
|
|
148
|
+
"openai": "^7.15.0",
|
|
121
149
|
"undici": "^7.29.0"
|
|
122
150
|
},
|
|
123
151
|
"lint-staged": {
|
|
@@ -125,9 +153,11 @@
|
|
|
125
153
|
},
|
|
126
154
|
"devDependencies": {
|
|
127
155
|
"gpt-tokenizer": "^4.0.0",
|
|
128
|
-
"lint-staged": "^17.
|
|
156
|
+
"lint-staged": "^17.5.1",
|
|
129
157
|
"release-it": "^21.0.2",
|
|
130
158
|
"standard": "^17.1.2",
|
|
159
|
+
"typescript": "^7.0.2",
|
|
131
160
|
"vitest": "^4.1.11"
|
|
132
|
-
}
|
|
161
|
+
},
|
|
162
|
+
"types": "./src/lib/index.d.ts"
|
|
133
163
|
}
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
// Short commands → noun + verb. Shared by dispatch and by completion, which
|
|
2
|
+
// must resolve the same chain the router would.
|
|
3
|
+
export const ALIASES = {
|
|
4
|
+
models: { noun: 'model', inject: ['list'] },
|
|
5
|
+
providers: { noun: 'provider', inject: ['list'] },
|
|
6
|
+
creators: { noun: 'creator', inject: ['list'] },
|
|
7
|
+
tags: { noun: 'tag', inject: ['list'] },
|
|
8
|
+
ls: { noun: 'model', inject: ['list'] },
|
|
9
|
+
show: { noun: 'model', inject: ['show'] },
|
|
10
|
+
search: { noun: 'model', inject: ['search'] },
|
|
11
|
+
stats: { noun: 'model', inject: ['stats'] },
|
|
12
|
+
check: { noun: 'model', inject: ['check'] },
|
|
13
|
+
setup: { noun: 'provider', inject: ['setup'] },
|
|
14
|
+
rank: { noun: 'model', inject: ['rank'] },
|
|
15
|
+
bench: { noun: 'model', inject: ['bench'] },
|
|
16
|
+
curate: { noun: 'model', inject: ['curate'] },
|
|
17
|
+
apply: { noun: 'model', inject: ['apply'] },
|
|
18
|
+
instructions: { noun: 'model', inject: ['instructions'] },
|
|
19
|
+
rl: { noun: 'ratelimit', inject: [] }
|
|
20
|
+
}
|
package/src/cli/ask.js
CHANGED
|
@@ -1,11 +1,28 @@
|
|
|
1
1
|
import mohdel, { silent } from '../lib/index.js'
|
|
2
|
-
import { loadDefaultEnv } from '../lib/common.js'
|
|
2
|
+
import { getConfig, loadDefaultEnv } from '../lib/common.js'
|
|
3
3
|
|
|
4
4
|
const noop = () => {}
|
|
5
5
|
|
|
6
6
|
// Friendly next-step hints for common ask-time failures. Pure pattern match on
|
|
7
7
|
// err.message — keeps the lib layer neutral, but gives CLI users a copy-pasteable
|
|
8
8
|
// command instead of just an error. Shared with `mo transcribe`.
|
|
9
|
+
// Waiting on a model is the one place mohdel has nothing to show for several
|
|
10
|
+
// seconds. The frames carry the resolved id, so the id is visible while it is
|
|
11
|
+
// useful and gone afterwards. stderr only, so pipes see nothing.
|
|
12
|
+
const startSpinner = (text) => {
|
|
13
|
+
if (!process.stderr.isTTY) return null
|
|
14
|
+
const frames = ['⠋', '⠙', '⠹', '⠸', '⠼', '⠴', '⠦', '⠧', '⠇', '⠏']
|
|
15
|
+
let i = 0
|
|
16
|
+
const timer = setInterval(() => {
|
|
17
|
+
process.stderr.write(`\r${frames[i++ % frames.length]} ${text}`)
|
|
18
|
+
}, 80)
|
|
19
|
+
timer.unref?.()
|
|
20
|
+
return () => {
|
|
21
|
+
clearInterval(timer)
|
|
22
|
+
process.stderr.write('\r\u001b[K')
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
|
|
9
26
|
export const hintsForError = (err, modelId) => {
|
|
10
27
|
const msg = String(err?.message || '')
|
|
11
28
|
const detail = String(err?.detail || '')
|
|
@@ -16,6 +33,7 @@ export const hintsForError = (err, modelId) => {
|
|
|
16
33
|
if (/not found in catalog/i.test(both)) {
|
|
17
34
|
if (provider) {
|
|
18
35
|
hints.push(`→ run: mo curate ${provider} # add upstream models from this provider`)
|
|
36
|
+
hints.push(`→ then: mo model instructions ${provider} # let your coding agent fill in the prices`)
|
|
19
37
|
hints.push(`→ or: mo model add ${modelId} # add this one manually`)
|
|
20
38
|
} else {
|
|
21
39
|
hints.push('→ run: mo ls # list available models')
|
|
@@ -49,18 +67,21 @@ Usage:
|
|
|
49
67
|
mo ask <model> "question" < file Combined: args + stdin
|
|
50
68
|
|
|
51
69
|
Options:
|
|
52
|
-
--effort <level> Thinking effort:
|
|
70
|
+
--effort <level> Thinking effort: none, low, medium, high, xhigh, max
|
|
53
71
|
--budget <tokens> Output token budget
|
|
54
72
|
--json Output full result as JSON
|
|
55
73
|
--stream Stream output to stdout in real time
|
|
56
|
-
-v, --verbose Show debug info on stderr (cooldown, rate
|
|
74
|
+
-v, --verbose Show debug info on stderr (cooldown, rate limits)
|
|
75
|
+
-q, --quiet Nothing on stderr but errors — no usage summary
|
|
57
76
|
|
|
58
77
|
Output:
|
|
59
78
|
stdout: model output text (raw, no formatting — or JSON with --json)
|
|
60
|
-
stderr:
|
|
79
|
+
stderr: token usage summary, and errors
|
|
80
|
+
--json omits the summary (the same numbers are in the payload);
|
|
81
|
+
--quiet omits it too, leaving stderr for failures alone
|
|
61
82
|
|
|
62
83
|
Examples:
|
|
63
|
-
mo ask
|
|
84
|
+
mo ask openai/gpt-5.6-luna "why is the sky blue"
|
|
64
85
|
cat article.txt | mo ask anthropic/claude-sonnet-4-6 "summarize this"
|
|
65
86
|
mo ask openai/gpt-5.4 --effort high "explain monads" --json | jq .cost`)
|
|
66
87
|
process.exit(0)
|
|
@@ -86,18 +107,23 @@ Examples:
|
|
|
86
107
|
const json = flag('--json')
|
|
87
108
|
const stream = flag('--stream')
|
|
88
109
|
const verbose = flag('--verbose') || flag('-v')
|
|
110
|
+
const quiet = flag('--quiet') || flag('-q')
|
|
89
111
|
const effort = flagVal('--effort')
|
|
90
112
|
const budget = flagVal('--budget')
|
|
91
113
|
|
|
92
114
|
// First remaining arg is model
|
|
93
|
-
|
|
115
|
+
// A model id always carries a provider segment, so a first argument without
|
|
116
|
+
// a slash is prompt text and the configured default model applies. Typing a
|
|
117
|
+
// full id on every call is the first thing that wears out.
|
|
118
|
+
const given = args[0]?.includes('/') ? args[0] : null
|
|
119
|
+
const modelId = given ?? (await getConfig()).defaultModel
|
|
94
120
|
if (!modelId) {
|
|
95
121
|
console.error('Usage: mo ask <model> [prompt]')
|
|
122
|
+
console.error('→ or: mo default # pick one, then "mo ask" needs no model')
|
|
96
123
|
process.exit(1)
|
|
97
124
|
}
|
|
98
125
|
|
|
99
|
-
|
|
100
|
-
const promptArgs = args.slice(1).join(' ').trim()
|
|
126
|
+
const promptArgs = args.slice(given ? 1 : 0).join(' ').trim()
|
|
101
127
|
|
|
102
128
|
// Read stdin if piped
|
|
103
129
|
let stdinContent = ''
|
|
@@ -140,15 +166,32 @@ Examples:
|
|
|
140
166
|
const options = {}
|
|
141
167
|
if (effort) options.outputEffort = effort
|
|
142
168
|
if (budget) options.outputBudget = parseInt(budget, 10)
|
|
169
|
+
|
|
170
|
+
// --verbose puts log lines on stderr on purpose; a spinner would be
|
|
171
|
+
// overwritten by them and leave its last frame behind.
|
|
172
|
+
const stopSpinner = (json || verbose || quiet) ? null : startSpinner(model.id)
|
|
173
|
+
let stopped = false
|
|
174
|
+
const stop = () => {
|
|
175
|
+
if (stopped) return
|
|
176
|
+
stopped = true
|
|
177
|
+
stopSpinner?.()
|
|
178
|
+
}
|
|
179
|
+
|
|
143
180
|
if (stream && !json) {
|
|
144
|
-
options.realtimeHandler = (delta) =>
|
|
181
|
+
options.realtimeHandler = (delta) => {
|
|
182
|
+
stop()
|
|
183
|
+
process.stdout.write(delta)
|
|
184
|
+
}
|
|
145
185
|
options.bufferOpts = { maxChars: 1, maxMs: 0 }
|
|
146
186
|
}
|
|
147
187
|
|
|
148
|
-
|
|
188
|
+
// Without a terminal the id is worth echoing only when it is not what was
|
|
189
|
+
// typed — an alias or a partial that resolved to something else.
|
|
190
|
+
if (!stopSpinner && !quiet && model.id !== modelId) process.stderr.write(`${model.id}\n`)
|
|
149
191
|
|
|
150
192
|
try {
|
|
151
193
|
const result = await model.answer(prompt, options)
|
|
194
|
+
stop()
|
|
152
195
|
const output = typeof result === 'string' ? result : result?.output || ''
|
|
153
196
|
const tokens = typeof result === 'object' ? result : {}
|
|
154
197
|
|
|
@@ -170,7 +213,7 @@ Examples:
|
|
|
170
213
|
if (output && !output.endsWith('\n')) process.stdout.write('\n')
|
|
171
214
|
}
|
|
172
215
|
|
|
173
|
-
// Token + timing summary to stderr
|
|
216
|
+
// Token + timing summary to stderr.
|
|
174
217
|
const summary = []
|
|
175
218
|
if (tokens.inputTokens) summary.push(`${tokens.inputTokens} in`)
|
|
176
219
|
if (tokens.outputTokens) summary.push(`${tokens.outputTokens} out`)
|
|
@@ -195,8 +238,11 @@ Examples:
|
|
|
195
238
|
if (ttft != null) summary.push(`${Math.round(ttft)}ms ttft`)
|
|
196
239
|
if (total != null) summary.push(`${Math.round(total)}ms total`)
|
|
197
240
|
}
|
|
198
|
-
|
|
241
|
+
// --json already carries these numbers in the payload; --quiet keeps
|
|
242
|
+
// stderr free so a caller can treat anything on it as a failure.
|
|
243
|
+
if (summary.length && !json && !quiet) process.stderr.write(`${summary.join(', ')}\n`)
|
|
199
244
|
} catch (err) {
|
|
245
|
+
stop()
|
|
200
246
|
console.error(`Error: ${err.detail || err.message}`)
|
|
201
247
|
for (const h of hintsForError(err, modelId)) console.error(h)
|
|
202
248
|
process.exit(1)
|
package/src/cli/backup.js
CHANGED
|
@@ -14,7 +14,8 @@ Usage:
|
|
|
14
14
|
model backup restore <slot> Restore from a backup slot
|
|
15
15
|
model backup diff <slot> Show changes between current and slot
|
|
16
16
|
|
|
17
|
-
Slots: prev (last save), daily (first save of the day),
|
|
17
|
+
Slots: prev (last save), daily (first save of the day),
|
|
18
|
+
weekly (first save of the week)`)
|
|
18
19
|
process.exit(0)
|
|
19
20
|
}
|
|
20
21
|
|
package/src/cli/check.js
CHANGED
|
@@ -1,106 +1,35 @@
|
|
|
1
1
|
import { label, err, warn, ok } from './colors.js'
|
|
2
|
-
import
|
|
3
|
-
import {
|
|
4
|
-
import {
|
|
5
|
-
import { getCuratedModels, loadDefaultEnv, catalogEntries, catalogValues } from '../lib/common.js'
|
|
6
|
-
|
|
7
|
-
// --- Local validation ---
|
|
8
|
-
|
|
9
|
-
const checkLocal = (curated) => {
|
|
10
|
-
const errors = []
|
|
11
|
-
const warnings = []
|
|
12
|
-
const knownProviders = new Set(Object.keys(providers))
|
|
13
|
-
|
|
14
|
-
for (const [key, spec] of catalogEntries(curated)) {
|
|
15
|
-
const [keyProvider] = key.split('/')
|
|
16
|
-
|
|
17
|
-
if (spec.deprecated) {
|
|
18
|
-
if (!curated[spec.deprecated]) {
|
|
19
|
-
errors.push(`${key}: deprecated target '${spec.deprecated}' not in curated`)
|
|
20
|
-
}
|
|
21
|
-
continue
|
|
22
|
-
}
|
|
23
|
-
|
|
24
|
-
for (const issue of validate(spec, key)) {
|
|
25
|
-
if (issue.severity === 'error') errors.push(`${key}: ${issue.field} — ${issue.message}`)
|
|
26
|
-
else warnings.push(`${key}: ${issue.field} — ${issue.message}`)
|
|
27
|
-
}
|
|
28
|
-
|
|
29
|
-
if (!knownProviders.has(keyProvider)) {
|
|
30
|
-
errors.push(`${key}: provider '${keyProvider}' not in providers.js`)
|
|
31
|
-
}
|
|
32
|
-
if (spec.provider && spec.provider !== keyProvider) {
|
|
33
|
-
errors.push(`${key}: spec.provider '${spec.provider}' doesn't match key prefix '${keyProvider}'`)
|
|
34
|
-
}
|
|
35
|
-
|
|
36
|
-
const providerConfig = providers[keyProvider]
|
|
37
|
-
if (providerConfig && spec.sdk && spec.sdk !== providerConfig.sdk) {
|
|
38
|
-
errors.push(`${key}: spec.sdk '${spec.sdk}' doesn't match provider sdk '${providerConfig.sdk}'`)
|
|
39
|
-
}
|
|
40
|
-
|
|
41
|
-
if (!spec.label) warnings.push(`${key}: missing label`)
|
|
42
|
-
|
|
43
|
-
for (const priceField of ['inputPrice', 'outputPrice', 'thinkingPrice']) {
|
|
44
|
-
const val = spec[priceField]
|
|
45
|
-
if (val != null && typeof val === 'object' && val.default == null) {
|
|
46
|
-
errors.push(`${key}: ${priceField} is tiered but missing 'default' key`)
|
|
47
|
-
}
|
|
48
|
-
}
|
|
49
|
-
|
|
50
|
-
if (spec.thinkingEffortLevels && !spec.defaultThinkingEffort) {
|
|
51
|
-
warnings.push(`${key}: has thinkingEffortLevels but no defaultThinkingEffort`)
|
|
52
|
-
}
|
|
53
|
-
|
|
54
|
-
const lanes = adapters[keyProvider]?.speedLanes
|
|
55
|
-
for (const [lane, overlay] of Object.entries(spec.speeds || {})) {
|
|
56
|
-
if (!lanes?.has(lane)) {
|
|
57
|
-
const detail = lanes ? `accepts: ${[...lanes].join(', ')}` : 'implements no speed lanes'
|
|
58
|
-
errors.push(`${key}: speeds.${lane} — provider '${keyProvider}' ${detail}; calls on that lane would fail at dispatch`)
|
|
59
|
-
}
|
|
60
|
-
for (const priceField of ['inputPrice', 'outputPrice', 'thinkingPrice']) {
|
|
61
|
-
const val = overlay[priceField]
|
|
62
|
-
if (val != null && typeof val === 'object' && val.default == null) {
|
|
63
|
-
errors.push(`${key}: speeds.${lane}.${priceField} is tiered but missing 'default' key`)
|
|
64
|
-
}
|
|
65
|
-
}
|
|
66
|
-
const priced = ['inputPrice', 'outputPrice'].some(f => overlay[f] != null)
|
|
67
|
-
if (!priced) {
|
|
68
|
-
warnings.push(`${key}: speeds.${lane} restates no prices — the lane will bill at base rates`)
|
|
69
|
-
}
|
|
70
|
-
}
|
|
71
|
-
|
|
72
|
-
if (Array.isArray(spec.tags)) {
|
|
73
|
-
for (const t of spec.tags) {
|
|
74
|
-
if (!isValidTag(t)) warnings.push(`${key}: invalid tag "${t}" — must match /^[a-zA-Z][a-zA-Z0-9._-]{0,31}$/`)
|
|
75
|
-
}
|
|
76
|
-
}
|
|
77
|
-
}
|
|
78
|
-
|
|
79
|
-
return { errors, warnings }
|
|
80
|
-
}
|
|
2
|
+
import { reviewCatalog } from '../lib/catalog-review.js'
|
|
3
|
+
import { localConventionsOrExit } from './local.js'
|
|
4
|
+
import { getCuratedModels, loadDefaultEnv, catalogValues } from '../lib/common.js'
|
|
81
5
|
|
|
82
6
|
// --- CLI ---
|
|
83
7
|
|
|
84
8
|
export async function runCheck (args) {
|
|
85
9
|
if (args.includes('-h') || args.includes('--help')) {
|
|
86
|
-
console.log(`mohdel model check — validate
|
|
10
|
+
console.log(`mohdel model check — validate the catalog
|
|
87
11
|
|
|
88
12
|
Usage:
|
|
89
13
|
model check [options]
|
|
14
|
+
model check --entry <file|-> Validate entries not yet in the catalog
|
|
90
15
|
|
|
91
16
|
Options:
|
|
92
17
|
--json Output as JSON
|
|
18
|
+
--entry <file|-> Read entries in curated.json shape and report what they
|
|
19
|
+
would change, without writing. 'mo model apply' writes.
|
|
93
20
|
|
|
94
21
|
Checks:
|
|
95
22
|
Schema types, required fields, deprecated targets, provider/sdk
|
|
96
|
-
consistency, tiered pricing, thinking config
|
|
97
|
-
|
|
98
|
-
Note: 0.90 drops the upstream-drift check that piggybacked on the
|
|
99
|
-
legacy per-provider SDK factory. If you need upstream drift
|
|
100
|
-
detection, file an issue — it'll be rebuilt on the /session stack.`)
|
|
23
|
+
consistency, tiered pricing, thinking config.`)
|
|
101
24
|
process.exit(0)
|
|
102
25
|
}
|
|
103
26
|
|
|
27
|
+
if (args.includes('--entry')) {
|
|
28
|
+
const { runCheckEntry } = await import('./entry.js')
|
|
29
|
+
await runCheckEntry(args)
|
|
30
|
+
return
|
|
31
|
+
}
|
|
32
|
+
|
|
104
33
|
loadDefaultEnv()
|
|
105
34
|
|
|
106
35
|
const json = args.includes('--json')
|
|
@@ -114,7 +43,7 @@ detection, file an issue — it'll be rebuilt on the /session stack.`)
|
|
|
114
43
|
console.log(`${label('Catalog:')} ${active} active, ${deprecated} deprecated\n`)
|
|
115
44
|
}
|
|
116
45
|
|
|
117
|
-
const { errors, warnings: localWarnings } =
|
|
46
|
+
const { errors, warnings: localWarnings } = reviewCatalog(curated, { local: await localConventionsOrExit() })
|
|
118
47
|
|
|
119
48
|
if (!json) {
|
|
120
49
|
if (errors.length) {
|