mohdel 3.1.0 → 3.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -1,6 +1,6 @@
1
1
  # Mohdel
2
2
 
3
- Self-hosted LLM gateway and SDK for Node — think LiteLLM, for the JS world. One `answer()` call for 13 providers or local inference; swap models by changing one string; get real per-call USD cost back on every result, with OpenTelemetry built in and process isolation when you need it. Your keys, your infra, no SaaS proxy in the path.
3
+ Self-hosted LLM gateway and SDK for Node — think LiteLLM, for the JS world. One `answer()` call for 14 providers or local inference; swap models by changing one string; get real per-call USD cost back on every result, with OpenTelemetry built in and process isolation when you need it. Your keys, your infra, no SaaS proxy in the path.
4
4
 
5
5
  ```bash
6
6
  npm install -g mohdel
@@ -20,7 +20,7 @@ curate novita` writes complete, priced entries on its own, and setup counts the
20
20
  models that cost nothing and offers to add all of them in one keystroke. A
21
21
  working catalog without a pricing page or a brief.
22
22
 
23
- Providers: Anthropic, OpenAI, Gemini, Mistral, Groq, xAI, Cerebras, Fireworks, DeepSeek, Qwen Cloud, Xiaomi, OpenRouter, Novita. Node 22+, ES modules.
23
+ Providers: Anthropic, OpenAI, Gemini, Mistral, Groq, xAI, Cerebras, Fireworks, DeepSeek, Qwen Cloud, Xiaomi, Meta Model API, OpenRouter, Novita. Node 22+, ES modules.
24
24
 
25
25
  Mohdel runs the inference layer of production stacks, among them [docAnalyzer](https://docanalyzer.ai), a document analysis and chat platform serving hundreds of thousands of users.
26
26
 
@@ -422,6 +422,7 @@ OPENROUTER_API_SK=sk-or-...
422
422
  NOVITA_API_SK=...
423
423
  QWEN_API_SK=sk-...
424
424
  XIAOMI_API_SK=...
425
+ META_API_SK=...
425
426
  COHERE_API_SK=...
426
427
  MOHDEL_LOCAL_API_SK=...
427
428
  ```
@@ -461,6 +462,7 @@ What each provider supports through mohdel's unified interface:
461
462
  | Mistral | Yes | Yes | Yes | No | No | `tool_choice: "any"` = required |
462
463
  | Qwen Cloud | Yes | Yes | No | No | Yes (`enable_thinking` + `thinking_budget`) | Alibaba DashScope intl; hybrid models think by default — effort `none` sends explicit off |
463
464
  | Xiaomi | Yes | Yes | Yes | No | Auto | MiMo; shared chat-completions path, `reasoning_content` captured |
465
+ | Meta | Yes | Yes | Yes | No | Yes (`reasoning.effort`) | OpenAI Responses API over `api.meta.ai/v1`, sent with `store: false`; Muse Spark reasoning cannot be turned off |
464
466
  | OpenRouter | Yes | Yes | Yes | No | Varies | Meta-provider; `providerOptions.openrouter` for routing prefs |
465
467
  | Local | Yes | Yes | Yes | No | No | Any OpenAI-compatible server; endpoint is the catalog entry's `baseURL` |
466
468
  | Cohere | n/a | n/a | n/a | n/a | n/a | Embeddings only: no chat models reach mohdel through it |
@@ -20,6 +20,7 @@ export const ADAPTER_NAMES = Object.freeze([
20
20
  'gemini',
21
21
  'groq',
22
22
  'local',
23
+ 'meta',
23
24
  'mistral',
24
25
  'novita',
25
26
  'openai',
@@ -20,6 +20,7 @@ import { fireworks } from './fireworks.js'
20
20
  import { gemini } from './gemini.js'
21
21
  import { groq } from './groq.js'
22
22
  import { local } from './local.js'
23
+ import { meta } from './meta.js'
23
24
  import { mistral } from './mistral.js'
24
25
  import { novita } from './novita.js'
25
26
  import { openai } from './openai.js'
@@ -38,6 +39,7 @@ export const adapters = Object.freeze({
38
39
  gemini,
39
40
  groq,
40
41
  local,
42
+ meta,
41
43
  mistral,
42
44
  novita,
43
45
  openai,
@@ -0,0 +1,29 @@
1
+ /**
2
+ * Meta Model API adapter — OpenAI Responses API over api.meta.ai/v1.
3
+ * Delegates to the `openai` adapter with a baseURL-configured client;
4
+ * the openai adapter branches on `providerOf(envelope.model)` for the
5
+ * fields that differ between vendors.
6
+ *
7
+ * @module session/adapters/meta
8
+ */
9
+
10
+ import OpenAI from 'openai'
11
+
12
+ import { openai } from './openai.js'
13
+ import { streamingDispatcher } from './_dispatcher.js'
14
+
15
+ const BASE_URL = 'https://api.meta.ai/v1'
16
+
17
+ /**
18
+ * @param {import('#core/envelope.js').CallEnvelope} envelope
19
+ * @param {{client?: any, signal?: AbortSignal}} [deps]
20
+ * @returns {AsyncGenerator<import('#core/events.js').Event>}
21
+ */
22
+ export async function * meta (envelope, deps = {}) {
23
+ const client = deps.client ?? new OpenAI({
24
+ apiKey: envelope.auth.key,
25
+ baseURL: BASE_URL,
26
+ fetchOptions: { dispatcher: streamingDispatcher() }
27
+ })
28
+ yield * openai(envelope, { ...deps, client })
29
+ }
@@ -360,6 +360,9 @@ function buildRequest (envelope, input, instructions) {
360
360
 
361
361
  if (envelope.speed) request.service_tier = envelope.speed
362
362
 
363
+ // Meta's Responses API stores every prompt and response unless told not to.
364
+ if (provider === 'meta') request.store = false
365
+
363
366
  return request
364
367
  }
365
368
 
package/js/session/run.js CHANGED
@@ -140,6 +140,16 @@ export async function * run (envelope, {
140
140
  return
141
141
  }
142
142
 
143
+ if (envelope.outputEffort) {
144
+ const effortErr = effortError(key, envelope.outputEffort, spec)
145
+ if (effortErr) {
146
+ log.warn({ provider, effort: envelope.outputEffort }, '[mohdel:answer] unsupported output effort')
147
+ endSpanError(span, new Error(effortErr.error.message))
148
+ yield effortErr
149
+ return
150
+ }
151
+ }
152
+
143
153
  if (envelope.speed) {
144
154
  const speedErr = speedError(key, envelope.speed, spec, provider, adapter)
145
155
  if (speedErr) {
@@ -337,29 +347,40 @@ export function normalizeModelId (envelope, resolveSpec) {
337
347
  return { envelope: next, key: base, spec: baseSpec }
338
348
  }
339
349
 
340
- if (!baseSpec.thinkingEffortLevels) {
341
- return {
342
- ...unresolved,
343
- error: errorEvent(
344
- `Model '${base}' does not support output effort (no thinkingEffortLevels). Cannot use ':${effort}' suffix.`,
345
- 'SESSION_INVALID_OUTPUT_EFFORT'
346
- )
347
- }
348
- }
349
- if (effort !== 'none' && !baseSpec.thinkingEffortLevels[effort]) {
350
- return {
351
- ...unresolved,
352
- error: errorEvent(
353
- `Model '${base}' does not support output effort level '${effort}'. Available: ${Object.keys(baseSpec.thinkingEffortLevels).join(', ')}`,
354
- 'SESSION_INVALID_OUTPUT_EFFORT'
355
- )
356
- }
357
- }
350
+ const effortErr = effortError(base, effort, baseSpec)
351
+ if (effortErr) return { ...unresolved, error: effortErr }
358
352
 
359
353
  next.outputEffort = effort
360
354
  return { envelope: next, key: base, spec: baseSpec }
361
355
  }
362
356
 
357
+ /**
358
+ * An effort level is valid only when the entry's `thinkingEffortLevels`
359
+ * declares it — `none` included, since some models cannot turn
360
+ * thinking off.
361
+ *
362
+ * @param {string} key
363
+ * @param {string} effort
364
+ * @param {any} spec
365
+ * @returns {import('#core/events.js').ErrorEvent | undefined}
366
+ */
367
+ export function effortError (key, effort, spec) {
368
+ const levels = spec.thinkingEffortLevels
369
+ if (!levels) {
370
+ return errorEvent(
371
+ `Model '${key}' does not support output effort (no thinkingEffortLevels). Cannot use effort '${effort}'.`,
372
+ 'SESSION_INVALID_OUTPUT_EFFORT'
373
+ )
374
+ }
375
+ if (!Object.hasOwn(levels, effort)) {
376
+ return errorEvent(
377
+ `Model '${key}' does not support output effort level '${effort}'. Available: ${Object.keys(levels).join(', ')}`,
378
+ 'SESSION_INVALID_OUTPUT_EFFORT'
379
+ )
380
+ }
381
+ return undefined
382
+ }
383
+
363
384
  /**
364
385
  * A `cache` marker is honoured on `text` parts only, with a TTL the
365
386
  * adapters know. The adapters would ignore any other marker, so it is
package/package.json CHANGED
@@ -1,12 +1,12 @@
1
1
  {
2
2
  "name": "mohdel",
3
- "version": "3.1.0",
3
+ "version": "3.2.0",
4
4
  "license": "MIT",
5
5
  "author": {
6
6
  "name": "Christophe Le Bars",
7
7
  "email": "clb@toort.net"
8
8
  },
9
- "description": "Self-hosted LLM gateway and SDK for Node — a LiteLLM-style unified API for 13 providers (Anthropic, OpenAI, Gemini, Mistral, Groq, xAI, DeepSeek, OpenRouter, …) with per-call USD cost tracking, streaming, tool calls, vision, speech-to-text, and built-in OpenTelemetry. Run in-process, or behind the process-isolated thin-gate for fault containment.",
9
+ "description": "Self-hosted LLM gateway and SDK for Node — a LiteLLM-style unified API for 14 providers (Anthropic, OpenAI, Gemini, Mistral, Groq, xAI, DeepSeek, OpenRouter, …) with per-call USD cost tracking, streaming, tool calls, vision, speech-to-text, and built-in OpenTelemetry. Run in-process, or behind the process-isolated thin-gate for fault containment.",
10
10
  "type": "module",
11
11
  "keywords": [
12
12
  "llm",
@@ -135,11 +135,11 @@
135
135
  "optionalDependencies": {
136
136
  "@opentelemetry/exporter-trace-otlp-grpc": "^0.222.0",
137
137
  "@opentelemetry/sdk-node": "^0.222.0",
138
- "chalk": "^6.0.0",
139
- "mohdel-thin-gate-linux-x64-gnu": "3.1.0"
138
+ "chalk": "^6.0.1",
139
+ "mohdel-thin-gate-linux-x64-gnu": "3.2.0"
140
140
  },
141
141
  "dependencies": {
142
- "@anthropic-ai/sdk": "^0.128.0",
142
+ "@anthropic-ai/sdk": "^0.129.0",
143
143
  "@cerebras/cerebras_cloud_sdk": "^1.91.0",
144
144
  "@clack/prompts": "^1.8.1",
145
145
  "@google/genai": "^2.24.0",
@@ -154,7 +154,7 @@
154
154
  },
155
155
  "devDependencies": {
156
156
  "gpt-tokenizer": "^4.0.0",
157
- "lint-staged": "^17.5.1",
157
+ "lint-staged": "^17.6.0",
158
158
  "release-it": "^21.0.3",
159
159
  "standard": "^17.1.2",
160
160
  "typescript": "^7.0.2",
@@ -36,7 +36,7 @@ const creators = {
36
36
  description: 'Kuaishou\'s KwaiPilot team builds KAT-Coder, a MoE coding model with strong agentic and multi-step reasoning for software engineering tasks.'
37
37
  },
38
38
  meta: {
39
- prefixes: ['llama', 'code-llama'],
39
+ prefixes: ['llama', 'code-llama', 'muse'],
40
40
  label: 'Meta',
41
41
  logo: 'meta.svg',
42
42
  description: 'Meta stewards the Llama ecosystem with open, widely adoptable models for chat, coding, and research.'
package/src/lib/index.js CHANGED
@@ -402,7 +402,7 @@ const mohdel = async ({ logger, verbosity: verbosityOpt, onSuccess, onFailure, c
402
402
  if (!modelSpec.thinkingEffortLevels) {
403
403
  throw new Error(`Model '${resolvedModelId}' does not support output effort (no thinkingEffortLevels). Cannot use ':${aliasOutputEffort}' suffix.`)
404
404
  }
405
- if (aliasOutputEffort !== 'none' && !modelSpec.thinkingEffortLevels[aliasOutputEffort]) {
405
+ if (!Object.hasOwn(modelSpec.thinkingEffortLevels, aliasOutputEffort)) {
406
406
  throw new Error(`Model '${resolvedModelId}' does not support output effort level '${aliasOutputEffort}'. Available: ${Object.keys(modelSpec.thinkingEffortLevels).join(', ')}`)
407
407
  }
408
408
  }
@@ -79,6 +79,13 @@ const PROVIDER_INFO = {
79
79
  hint: 'Create an API key at novita.ai → Dashboard → API Key',
80
80
  free: false
81
81
  },
82
+ meta: {
83
+ label: 'Meta Model API',
84
+ description: 'Muse Spark — reasoning, long context, image, video and PDF input.',
85
+ url: 'https://dev.meta.ai/',
86
+ hint: 'Create an API key in the Meta Model API console at dev.meta.ai',
87
+ free: false
88
+ },
82
89
  xiaomi: {
83
90
  label: 'Xiaomi MiMo',
84
91
  description: 'MiMo — vision and text models.',
@@ -86,6 +86,18 @@ const providers = {
86
86
  contextSemantics: 'shared',
87
87
  outputCapStrategy: 'accept'
88
88
  },
89
+ meta: {
90
+ sdk: 'openai',
91
+ apiKeyEnv: 'META_API_SK',
92
+ baseURL: 'https://api.meta.ai/v1',
93
+ createConfiguration: apiKey => ({ apiKey }),
94
+ references: {
95
+ pricing: 'https://dev.meta.ai/docs/pricing-rate-limits',
96
+ models: 'https://dev.meta.ai/docs/models',
97
+ rateLimits: 'https://dev.meta.ai/docs/pricing-rate-limits'
98
+ },
99
+ outputCapStrategy: 'accept'
100
+ },
89
101
  cohere: {
90
102
  sdk: 'cohere',
91
103
  api: 'embeddings',