mohdel 0.121.2 → 0.122.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -1,6 +1,6 @@
1
1
  # Mohdel
2
2
 
3
- Self-hosted LLM gateway and SDK for Node — think LiteLLM, for the JS world. One `answer()` call for 13 providers; swap models by changing one string; get real per-call USD cost back on every result, with OpenTelemetry built in and process isolation when you need it. Your keys, your infra, no SaaS proxy in the path.
3
+ Self-hosted LLM gateway and SDK for Node — think LiteLLM, for the JS world. One `answer()` call for 13 providers or local inference; swap models by changing one string; get real per-call USD cost back on every result, with OpenTelemetry built in and process isolation when you need it. Your keys, your infra, no SaaS proxy in the path.
4
4
 
5
5
  ```bash
6
6
  npm install -g mohdel
@@ -279,10 +279,13 @@ FIREWORKS_API_SK=fw_...
279
279
  DEEPSEEK_API_SK=sk-...
280
280
  OPENROUTER_API_SK=sk-or-...
281
281
  NOVITA_API_SK=...
282
+ MOHDEL_LOCAL_API_SK=...
282
283
  ```
283
284
 
284
285
  Only set keys for providers you use. Run `mo` with no arguments for interactive setup.
285
286
 
287
+ `local/` routes each model to the `baseURL` of its catalog entry — any OpenAI-compatible chat-completions server (Ollama, vLLM, llama.cpp server, LM Studio). There is no default endpoint and no per-call override: an entry without `baseURL` fails `mo model check`, a call to it fails with `CONFIGURATION_MISSING`, so a call never falls through to a cloud provider. `MOHDEL_LOCAL_API_SK` is an optional bearer token. Entry format in [docs/CATALOG.md](docs/CATALOG.md#self-hosted-local-entries).
288
+
286
289
  ### File locations
287
290
 
288
291
  | Path | Purpose |
@@ -314,6 +317,7 @@ What each provider supports through mohdel's unified interface:
314
317
  | Qwen Cloud | No | Yes | No | No | Yes (`enable_thinking` + `thinking_budget`) | Alibaba DashScope intl; hybrid models think by default — effort `none` sends explicit off |
315
318
  | Xiaomi | No | Yes | Yes | No | Auto | MiMo; shared chat-completions path, `reasoning_content` captured |
316
319
  | OpenRouter | Yes | Yes | Yes | No | Varies | Meta-provider; `providerOptions.openrouter` for routing prefs |
320
+ | Local | Yes | Yes | Yes | No | No | Any OpenAI-compatible server; endpoint is the catalog entry's `baseURL` |
317
321
  | Novita | No | No | No | No | No | Image generation only |
318
322
 
319
323
  Adapter capability ≠ model capability — whether a given model accepts images, tools, or thinking effort depends on the model spec in `curated.json`. The adapter passes through what the envelope supplies; the provider rejects unsupported combos.
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "$schema": "./curated.schema.json",
3
- "_comment": "Worked examples for ~/.config/mohdel/curated.json. See docs/CATALOG.md for the field reference. Each top-level key is '<provider>/<model-id>'. The provider segment is the routing key; the model-id segment must match the value of the 'model' field below (which is the literal id mohdel sends to the provider's API).",
3
+ "_comment": "Worked examples for ~/.config/mohdel/curated.json. See docs/CATALOG.md for the field reference. Each top-level key is '<provider>/<model-id>'. The provider segment is the routing key; the 'model' field is the literal id mohdel sends to the provider's API and may differ from the model-id segment.",
4
4
 
5
5
  "anthropic/claude-haiku-4-5": {
6
6
  "_comment_a": "Minimal-but-useful entry. Required fields only: model, creator, inputFormat. Everything else is recommended — without prices, 'cost' on results will be 0; without contextTokenLimit, callers can't enforce input budgets.",
@@ -89,5 +89,17 @@
89
89
  "tpmLimit": 2000000,
90
90
  "rateLimitScope": "model",
91
91
  "tags": ["chat", "cheap"]
92
+ },
93
+
94
+ "local/llama3.1-8b": {
95
+ "_comment_f": "Self-hosted entry: 'baseURL' names the OpenAI-compatible server (required on local/ entries, rejected elsewhere). The key carries no ':' (mohdel reads ':' as the effort suffix); 'model' carries the server's own tag. No prices, so cost is 0.",
96
+ "model": "llama3.1:8b",
97
+ "baseURL": "http://127.0.0.1:11434/v1",
98
+ "creator": "meta",
99
+ "provider": "local",
100
+ "sdk": "openai",
101
+ "label": "Llama 3.1 8B (local)",
102
+ "inputFormat": ["text"],
103
+ "tags": ["local"]
92
104
  }
93
105
  }
@@ -70,6 +70,10 @@
70
70
  "minItems": 1,
71
71
  "uniqueItems": true
72
72
  },
73
+ "baseURL": {
74
+ "type": "string",
75
+ "description": "local/ entries only: the OpenAI-compatible server this model is served from (e.g. http://127.0.0.1:11434/v1). Required there, rejected elsewhere."
76
+ },
73
77
  "provider": {
74
78
  "type": "string",
75
79
  "description": "Routing provider. Defaults to the provider segment of the catalog key."
@@ -87,10 +87,6 @@
87
87
  /**
88
88
  * @typedef {object} Auth
89
89
  * @property {string} key Provider API key. Redact in logs; never persist.
90
- * @property {string} [baseURL]
91
- * Optional override of the adapter's default provider endpoint.
92
- * Lets callers point at a self-hosted deployment, regional endpoint,
93
- * proxy, or test server. Adapters treat it as `baseURL ?? ADAPTER_DEFAULT`.
94
90
  */
95
91
 
96
92
  /**
@@ -271,7 +271,7 @@ function newCallId () {
271
271
 
272
272
  /**
273
273
  * Turn the factory's `configuration` bag into an envelope `auth` object.
274
- * Accepts `apiKey` + `baseURL`; anything else is rejected with
274
+ * Accepts `apiKey`; anything else is rejected with
275
275
  * `CONFIGURATION_UNSUPPORTED` rather than silently dropped — otherwise
276
276
  * a caller could believe their proxy / org header / timeout is in
277
277
  * effect when it isn't, which in the worst case leaks auth keys and
@@ -280,7 +280,7 @@ function newCallId () {
280
280
  * @param {any} configuration
281
281
  * @returns {import('#core/envelope.js').Auth}
282
282
  */
283
- const ALLOWED_CONFIG_KEYS = new Set(['apiKey', 'baseURL'])
283
+ const ALLOWED_CONFIG_KEYS = new Set(['apiKey'])
284
284
 
285
285
  export function configToAuth (configuration) {
286
286
  if (!configuration) return { key: '' }
@@ -289,15 +289,13 @@ export function configToAuth (configuration) {
289
289
  throw new MohdelError('unsupported per-call configuration', {
290
290
  type: 'CONFIGURATION_UNSUPPORTED',
291
291
  detail:
292
- 'per-call SDK configuration is limited to `apiKey` and `baseURL`. ' +
292
+ 'per-call SDK configuration is limited to `apiKey`. ' +
293
293
  `Unsupported keys: ${unsupported.join(', ')}. ` +
294
294
  'Move other fields to environment variables or pin them at factory construction.',
295
295
  retryable: false
296
296
  })
297
297
  }
298
- const auth = { key: configuration.apiKey || '' }
299
- if (configuration.baseURL) auth.baseURL = configuration.baseURL
300
- return auth
298
+ return { key: configuration.apiKey || '' }
301
299
  }
302
300
 
303
301
  /**
@@ -23,7 +23,7 @@ const BASE_URL = 'https://api.deepseek.com'
23
23
  export async function * deepseek (envelope, deps = {}) {
24
24
  const client = deps.client ?? new OpenAI({
25
25
  apiKey: envelope.auth.key,
26
- baseURL: envelope.auth.baseURL || BASE_URL,
26
+ baseURL: BASE_URL,
27
27
  fetchOptions: { dispatcher: streamingDispatcher() }
28
28
  })
29
29
  yield * runChatCompletions(envelope, client, {
@@ -28,7 +28,7 @@ const BASE_URL = 'https://api.fireworks.ai/inference/v1'
28
28
  export async function * fireworks (envelope, deps = {}) {
29
29
  const client = deps.client ?? new OpenAI({
30
30
  apiKey: envelope.auth.key,
31
- baseURL: envelope.auth.baseURL || BASE_URL,
31
+ baseURL: BASE_URL,
32
32
  fetchOptions: { dispatcher: streamingDispatcher() }
33
33
  })
34
34
  yield * runChatCompletions(envelope, client, {
@@ -19,6 +19,7 @@ import { fake } from './fake.js'
19
19
  import { fireworks } from './fireworks.js'
20
20
  import { gemini } from './gemini.js'
21
21
  import { groq } from './groq.js'
22
+ import { local } from './local.js'
22
23
  import { mistral } from './mistral.js'
23
24
  import { novita } from './novita.js'
24
25
  import { openai } from './openai.js'
@@ -36,6 +37,7 @@ export const adapters = Object.freeze({
36
37
  fireworks,
37
38
  gemini,
38
39
  groq,
40
+ local,
39
41
  mistral,
40
42
  novita,
41
43
  openai,
@@ -0,0 +1,60 @@
1
+ /**
2
+ * Local adapter — any OpenAI-compatible chat-completions server
3
+ * (Ollama, vLLM, llama.cpp server, LM Studio). The endpoint is the
4
+ * catalog entry's `baseURL`; there is no default and no per-call
5
+ * override, so a call can never fall through to a cloud provider.
6
+ *
7
+ * @module session/adapters/local
8
+ */
9
+
10
+ import OpenAI from 'openai'
11
+
12
+ import { catalogKey } from '#core/model-id.js'
13
+
14
+ import { getSpec } from './_catalog.js'
15
+ import { runChatCompletions } from './_chat_completions.js'
16
+ import { streamingDispatcher } from './_dispatcher.js'
17
+
18
+ // The SDK constructor rejects an empty `apiKey`. An unauthenticated
19
+ // server gets a placeholder key that never reaches the wire: the
20
+ // explicit `Authorization: null` default header makes the SDK omit
21
+ // the header entirely.
22
+ const UNAUTHENTICATED = 'unauthenticated'
23
+
24
+ /**
25
+ * @param {import('#core/envelope.js').CallEnvelope} envelope
26
+ * @param {{client?: any, signal?: AbortSignal, log?: any, span?: any}} [deps]
27
+ * @returns {AsyncGenerator<import('#core/events.js').Event>}
28
+ */
29
+ export async function * local (envelope, deps = {}) {
30
+ const key = catalogKey(envelope.model)
31
+ const baseURL = getSpec(key)?.baseURL
32
+ if (!baseURL) {
33
+ yield {
34
+ type: 'error',
35
+ error: {
36
+ message: 'local provider has no endpoint',
37
+ detail: `catalog entry '${key}' has no baseURL`,
38
+ severity: 'error',
39
+ retryable: false,
40
+ type: 'CONFIGURATION_MISSING'
41
+ }
42
+ }
43
+ return
44
+ }
45
+ const client = deps.client ?? new OpenAI({
46
+ baseURL,
47
+ fetchOptions: { dispatcher: streamingDispatcher() },
48
+ ...(envelope.auth.key
49
+ ? { apiKey: envelope.auth.key }
50
+ : { apiKey: UNAUTHENTICATED, defaultHeaders: { Authorization: null } })
51
+ })
52
+ yield * runChatCompletions(envelope, client, {
53
+ provider: 'local',
54
+ stream: true
55
+ }, {
56
+ signal: deps.signal,
57
+ log: deps.log,
58
+ span: deps.span
59
+ })
60
+ }
@@ -22,7 +22,7 @@ const BASE_URL = 'https://api.mistral.ai/v1'
22
22
  export async function * mistral (envelope, deps = {}) {
23
23
  const client = deps.client ?? new OpenAI({
24
24
  apiKey: envelope.auth.key,
25
- baseURL: envelope.auth.baseURL || BASE_URL,
25
+ baseURL: BASE_URL,
26
26
  fetchOptions: { dispatcher: streamingDispatcher() }
27
27
  })
28
28
  yield * runChatCompletions(envelope, client, {
@@ -18,7 +18,7 @@ const BASE_URL = 'https://api.novita.ai/openai'
18
18
  * @returns {AsyncGenerator<import('#core/events.js').Event>}
19
19
  */
20
20
  export async function * novita (envelope, deps = {}) {
21
- const client = deps.client ?? new OpenAI({ apiKey: envelope.auth.key, baseURL: envelope.auth.baseURL || BASE_URL })
21
+ const client = deps.client ?? new OpenAI({ apiKey: envelope.auth.key, baseURL: BASE_URL })
22
22
  yield * runChatCompletions(envelope, client, {
23
23
  provider: 'novita'
24
24
  }, {
@@ -46,7 +46,6 @@ import { streamingDispatcher } from './_dispatcher.js'
46
46
  export async function * openai (envelope, deps = {}) {
47
47
  const client = deps.client ?? new OpenAI({
48
48
  apiKey: envelope.auth.key,
49
- ...(envelope.auth.baseURL ? { baseURL: envelope.auth.baseURL } : {}),
50
49
  fetchOptions: { dispatcher: streamingDispatcher() }
51
50
  })
52
51
  const signal = deps.signal
@@ -47,7 +47,7 @@ export async function * openrouter (envelope, deps = {}) {
47
47
  }
48
48
  const client = deps.client ?? new OpenAI({
49
49
  apiKey: envelope.auth.key,
50
- baseURL: envelope.auth.baseURL || BASE_URL,
50
+ baseURL: BASE_URL,
51
51
  defaultHeaders,
52
52
  fetchOptions: { dispatcher: streamingDispatcher() }
53
53
  })
@@ -23,7 +23,7 @@ const BASE_URL = 'https://dashscope-intl.aliyuncs.com/compatible-mode/v1'
23
23
  export async function * qwen (envelope, deps = {}) {
24
24
  const client = deps.client ?? new OpenAI({
25
25
  apiKey: envelope.auth.key,
26
- baseURL: envelope.auth.baseURL || BASE_URL,
26
+ baseURL: BASE_URL,
27
27
  fetchOptions: { dispatcher: streamingDispatcher() }
28
28
  })
29
29
  yield * runChatCompletions(envelope, client, {
@@ -48,7 +48,7 @@ export function createTranscriptionAdapter ({ baseURL, responseFormat }) {
48
48
  if (envelope.language) form.append('language', envelope.language)
49
49
  if (envelope.prompt) form.append('prompt', envelope.prompt)
50
50
 
51
- const root = (envelope.auth.baseURL || baseURL).replace(/\/$/, '')
51
+ const root = baseURL.replace(/\/$/, '')
52
52
  let res
53
53
  try {
54
54
  res = await fetchFn(`${root}/audio/transcriptions`, {
@@ -22,7 +22,7 @@ const BASE_URL = 'https://api.x.ai/v1'
22
22
  export async function * xai (envelope, deps = {}) {
23
23
  const client = deps.client ?? new OpenAI({
24
24
  apiKey: envelope.auth.key,
25
- baseURL: envelope.auth.baseURL || BASE_URL,
25
+ baseURL: BASE_URL,
26
26
  fetchOptions: { dispatcher: streamingDispatcher() }
27
27
  })
28
28
  yield * openai(envelope, { ...deps, client })
@@ -22,7 +22,7 @@ const BASE_URL = 'https://api.xiaomimimo.com/v1'
22
22
  export async function * xiaomi (envelope, deps = {}) {
23
23
  const client = deps.client ?? new OpenAI({
24
24
  apiKey: envelope.auth.key,
25
- baseURL: envelope.auth.baseURL || BASE_URL,
25
+ baseURL: BASE_URL,
26
26
  fetchOptions: { dispatcher: streamingDispatcher() }
27
27
  })
28
28
  yield * runChatCompletions(envelope, client, {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mohdel",
3
- "version": "0.121.2",
3
+ "version": "0.122.0",
4
4
  "license": "MIT",
5
5
  "author": {
6
6
  "name": "Christophe Le Bars",
@@ -108,7 +108,7 @@
108
108
  "@opentelemetry/exporter-trace-otlp-grpc": "^0.222.0",
109
109
  "@opentelemetry/sdk-node": "^0.222.0",
110
110
  "chalk": "^6.0.0",
111
- "mohdel-thin-gate-linux-x64-gnu": "0.121.2"
111
+ "mohdel-thin-gate-linux-x64-gnu": "0.122.0"
112
112
  },
113
113
  "dependencies": {
114
114
  "@anthropic-ai/sdk": "^0.122.0",
package/src/cli/doctor.js CHANGED
@@ -51,6 +51,7 @@ Exit code:
51
51
 
52
52
  // 2. API keys per provider
53
53
  for (const [name, def] of Object.entries(providers)) {
54
+ if (!def.apiKeyEnv) continue
54
55
  if (getAPIKey(def.apiKeyEnv)) {
55
56
  report.keys.configured.push({ provider: name, envVar: def.apiKeyEnv })
56
57
  } else {
@@ -137,7 +138,7 @@ Exit code:
137
138
  }
138
139
 
139
140
  console.log()
140
- console.log(label(`API keys (${report.keys.configured.length} of ${Object.keys(providers).length})`))
141
+ console.log(label(`API keys (${report.keys.configured.length} of ${Object.values(providers).filter(def => def.apiKeyEnv).length})`))
141
142
  for (const k of report.keys.configured) {
142
143
  console.log(row(ok('✓'), k.provider, k.envVar))
143
144
  }
@@ -19,7 +19,8 @@ const providers = {
19
19
  sdk: 'openai',
20
20
  api: 'chatCompletions',
21
21
  apiKeyEnv: 'DEEPSEEK_API_SK',
22
- createConfiguration: apiKey => ({ baseURL: 'https://api.deepseek.com', apiKey }),
22
+ baseURL: 'https://api.deepseek.com',
23
+ createConfiguration: apiKey => ({ apiKey }),
23
24
  creators: ['deepseek'],
24
25
  contextSemantics: 'shared',
25
26
  outputCapStrategy: 'accept'
@@ -27,7 +28,8 @@ const providers = {
27
28
  fireworks: {
28
29
  sdk: 'fireworks',
29
30
  apiKeyEnv: 'FIREWORKS_API_SK',
30
- createConfiguration: apiKey => ({ apiKey, baseURL: 'https://api.fireworks.ai/inference/v1' }),
31
+ baseURL: 'https://api.fireworks.ai/inference/v1',
32
+ createConfiguration: apiKey => ({ apiKey }),
31
33
  creators: ['meta', 'alibaba'],
32
34
  contextSemantics: 'shared',
33
35
  outputCapStrategy: 'accept'
@@ -46,11 +48,21 @@ const providers = {
46
48
  createConfiguration: apiKey => ({ apiKey }),
47
49
  creators: ['meta']
48
50
  },
51
+ local: {
52
+ sdk: 'openai',
53
+ api: 'chatCompletions',
54
+ catalog: false,
55
+ resolveConfiguration: () => ({ apiKey: process.env.MOHDEL_LOCAL_API_SK || '' }),
56
+ creators: [],
57
+ contextSemantics: 'shared',
58
+ outputCapStrategy: 'accept'
59
+ },
49
60
  mistral: {
50
61
  sdk: 'openai',
51
62
  api: 'chatCompletions',
52
63
  apiKeyEnv: 'MISTRAL_API_SK',
53
- createConfiguration: apiKey => ({ baseURL: 'https://api.mistral.ai/v1', apiKey }),
64
+ baseURL: 'https://api.mistral.ai/v1',
65
+ createConfiguration: apiKey => ({ apiKey }),
54
66
  creators: ['mistral']
55
67
  },
56
68
  novita: {
@@ -58,7 +70,8 @@ const providers = {
58
70
  api: 'chatCompletions',
59
71
  imageHandler: 'novita',
60
72
  apiKeyEnv: 'NOVITA_API_SK',
61
- createConfiguration: apiKey => ({ apiKey, baseURL: 'https://api.novita.ai/openai' }),
73
+ baseURL: 'https://api.novita.ai/openai',
74
+ createConfiguration: apiKey => ({ apiKey }),
62
75
  creators: ['deepseek', 'openai', 'bfl'],
63
76
  contextSemantics: 'shared',
64
77
  outputCapStrategy: 'error'
@@ -74,14 +87,16 @@ const providers = {
74
87
  openrouter: {
75
88
  sdk: 'openrouter',
76
89
  apiKeyEnv: 'OPENROUTER_API_SK',
77
- createConfiguration: apiKey => ({ baseURL: 'https://openrouter.ai/api/v1', apiKey }),
90
+ baseURL: 'https://openrouter.ai/api/v1',
91
+ createConfiguration: apiKey => ({ apiKey }),
78
92
  creators: []
79
93
  },
80
94
  qwen: {
81
95
  sdk: 'openai',
82
96
  api: 'chatCompletions',
83
97
  apiKeyEnv: 'QWEN_API_SK',
84
- createConfiguration: apiKey => ({ baseURL: 'https://dashscope-intl.aliyuncs.com/compatible-mode/v1', apiKey }),
98
+ baseURL: 'https://dashscope-intl.aliyuncs.com/compatible-mode/v1',
99
+ createConfiguration: apiKey => ({ apiKey }),
85
100
  creators: ['alibaba'],
86
101
  contextSemantics: 'shared',
87
102
  outputCapStrategy: 'accept'
@@ -89,7 +104,8 @@ const providers = {
89
104
  xai: {
90
105
  sdk: 'openai',
91
106
  apiKeyEnv: 'XAI_API_SK',
92
- createConfiguration: apiKey => ({ baseURL: 'https://api.x.ai/v1', apiKey }),
107
+ baseURL: 'https://api.x.ai/v1',
108
+ createConfiguration: apiKey => ({ apiKey }),
93
109
  creators: ['xai'],
94
110
  contextSemantics: 'shared',
95
111
  outputCapStrategy: 'accept'
@@ -98,7 +114,8 @@ const providers = {
98
114
  sdk: 'openai',
99
115
  api: 'chatCompletions',
100
116
  apiKeyEnv: 'XIAOMI_API_SK',
101
- createConfiguration: apiKey => ({ baseURL: 'https://api.xiaomimimo.com/v1', apiKey }),
117
+ baseURL: 'https://api.xiaomimimo.com/v1',
118
+ createConfiguration: apiKey => ({ apiKey }),
102
119
  creators: ['xiaomi'],
103
120
  contextSemantics: 'shared',
104
121
  outputCapStrategy: 'accept'
package/src/lib/schema.js CHANGED
@@ -17,6 +17,7 @@ const validateSpeeds = (speeds) => {
17
17
 
18
18
  const fieldDefs = {
19
19
  model: { type: 'string', required: true },
20
+ baseURL: { type: 'string' },
20
21
  provider: { type: 'string' },
21
22
  sdk: { type: 'string' },
22
23
  type: { type: 'string', default: 'model' },
@@ -122,6 +123,14 @@ const validate = (entry, curatedKey, { strict = false } = {}) => {
122
123
  }
123
124
  }
124
125
 
126
+ const provider = entry.provider ?? (typeof curatedKey === 'string' ? curatedKey.split('/')[0] : undefined)
127
+ if (provider === 'local' && !isDeprecatedStub && !entry.baseURL) {
128
+ issues.push({ field: 'baseURL', message: 'required for local/ entries', severity: 'error' })
129
+ }
130
+ if (provider !== 'local' && entry.baseURL !== undefined) {
131
+ issues.push({ field: 'baseURL', message: 'only local/ entries carry an endpoint', severity: 'error' })
132
+ }
133
+
125
134
  if (strict) {
126
135
  for (const key of Object.keys(entry)) {
127
136
  if (!knownFields.has(key) && !COMPUTED_FIELDS.has(key)) {
package/src/lib/select.js CHANGED
@@ -41,7 +41,7 @@ export const initializeAPIs = async () => {
41
41
  const sdkPath = `./catalog/${config.sdk}.js`
42
42
  const { default: API } = await import(sdkPath)
43
43
 
44
- api[name] = API(sdkConfig, {}, silent)
44
+ api[name] = API({ ...sdkConfig, baseURL: config.baseURL }, {}, silent)
45
45
  providersWithKeys.push(name)
46
46
  } catch (err) {
47
47
  console.error(`Error initializing provider ${name} api:`, err.message)