mohdel 0.124.0 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +155 -30
- package/config/curated.schema.json +32 -10
- package/js/client/call.js +76 -19
- package/js/client/gate-binary.js +5 -0
- package/js/client/index.js +1 -0
- package/js/core/envelope.js +5 -1
- package/js/factory/bridge.js +2 -2
- package/js/session/adapters/_cancelled.js +0 -6
- package/js/session/adapters/_chat_completions.js +25 -0
- package/js/session/adapters/_output_cap.js +30 -0
- package/js/session/adapters/_registry.js +44 -0
- package/js/session/adapters/anthropic.js +8 -5
- package/js/session/adapters/gemini.js +4 -0
- package/js/session/adapters/openai.js +2 -1
- package/js/session/run.js +14 -4
- package/js/session/run_image.js +12 -3
- package/package.json +49 -19
- package/src/cli/aliases.js +20 -0
- package/src/cli/ask.js +58 -12
- package/src/cli/backup.js +2 -1
- package/src/cli/check.js +15 -86
- package/src/cli/complete.js +130 -0
- package/src/cli/default.js +34 -13
- package/src/cli/doctor.js +33 -14
- package/src/cli/entry.js +173 -0
- package/src/cli/index.js +77 -66
- package/src/cli/instructions.js +349 -0
- package/src/cli/local.js +14 -0
- package/src/cli/model.js +184 -37
- package/src/cli/onboard.js +186 -121
- package/src/cli/rank.js +2 -1
- package/src/cli/ratelimit.js +3 -3
- package/src/cli/tag.js +2 -0
- package/src/lib/assistants.js +93 -0
- package/src/lib/catalog/openrouter.js +6 -1
- package/src/lib/catalog-review.js +195 -0
- package/src/lib/common.js +14 -1
- package/src/lib/creators.js +35 -0
- package/src/lib/index.js +17 -3
- package/src/lib/local-conventions.js +120 -0
- package/src/lib/provider-info.js +98 -0
- package/src/lib/providers.js +69 -14
- package/src/lib/schema.js +15 -3
- package/src/lib/select.js +125 -67
- package/js/session/adapters/image/index.js +0 -40
package/README.md
CHANGED
|
@@ -4,26 +4,42 @@ Self-hosted LLM gateway and SDK for Node — think LiteLLM, for the JS world. On
|
|
|
4
4
|
|
|
5
5
|
```bash
|
|
6
6
|
npm install -g mohdel
|
|
7
|
-
mo
|
|
8
|
-
mo
|
|
7
|
+
mo # pick a provider, paste your key, pull its models
|
|
8
|
+
mo model instructions openai > mohdel-brief.md # prices live on a docs page — hand it to your agent
|
|
9
|
+
mo ask openai/gpt-5.6-luna "why is the sky blue"
|
|
9
10
|
```
|
|
10
11
|
|
|
12
|
+
Almost no provider API returns prices, context limits or thinking budgets.
|
|
13
|
+
They live on a docs page, so mohdel writes a brief and the coding agent you
|
|
14
|
+
already run reads the page and drafts the entries. `mo` offers this at the end
|
|
15
|
+
of setup. Nothing runs on your key but that agent.
|
|
16
|
+
|
|
17
|
+
**No coding agent?** OpenRouter is the exception — it publishes per-token
|
|
18
|
+
prices in its own model list, so mohdel can read them. Setup counts the models
|
|
19
|
+
that cost nothing and offers to add all of them in one keystroke; `mo curate
|
|
20
|
+
openrouter` writes complete, priced entries for the paid ones. Free tier, no
|
|
21
|
+
card, and a working catalog without a pricing page or a brief.
|
|
22
|
+
|
|
11
23
|
Providers: Anthropic, OpenAI, Gemini, Mistral, Groq, xAI, Cerebras, Fireworks, DeepSeek, Qwen Cloud, Xiaomi, OpenRouter, Novita. Node 22+, ES modules.
|
|
12
24
|
|
|
25
|
+
Mohdel runs the inference layer of production stacks, among them [docAnalyzer](https://docanalyzer.ai), a document analysis and chat platform serving hundreds of thousands of users.
|
|
26
|
+
|
|
13
27
|
## Why mohdel
|
|
14
28
|
|
|
15
|
-
- **Real numbers on every call.** Token counts and per-call USD cost computed from your own pricing catalog (`curated.json`) — not estimates, not provider-specific shapes. Bill tenants, alert on spend, reconcile invoices. See [docs/CATALOG.md](docs/CATALOG.md)
|
|
29
|
+
- **Real numbers on every call.** Token counts and per-call USD cost computed from your own pricing catalog (`curated.json`) — not estimates, not provider-specific shapes. Bill tenants, alert on spend, reconcile invoices. Your own catalog means your negotiated rates and your own tags, and it is not a spreadsheet you maintain: `mo model instructions` hands the provider's docs page to your coding agent, which drafts the entries for you to review. See [docs/CATALOG.md](docs/CATALOG.md).
|
|
16
30
|
- **One interface across providers.** Same `answer()` call, same event stream, same `{ status, output, inputTokens, outputTokens, cost }` result. Switching from `anthropic/claude-sonnet-4-6` to `openai/gpt-5.4-mini` is one string change — adapter differences stay inside mohdel.
|
|
17
31
|
- **Self-hosted, no vendor in the path.** API keys live in `~/.config/mohdel/`. Mohdel calls provider APIs directly; nothing routes through a third party, nothing marks up your tokens, no extra hop of availability risk.
|
|
32
|
+
- **Nothing to compromise.** No network listener, no credential store, no tool execution. Mohdel runs a model call and returns the result; it cannot read a file, run a command, or hand back a key. See [Attack surface](#attack-surface).
|
|
18
33
|
- **Observability without instrumentation.** OpenTelemetry spans, trace-linked logs, and OTLP metrics over one endpoint. Set `OTEL_EXPORTER_OTLP_ENDPOINT`; everything else is wired.
|
|
34
|
+
- **Fully typed.** Declarations are generated from the source's own JSDoc and ship with the package — `CallEnvelope`, `Event`, `AnswerResult` and `MohdelError` are the frozen wire contract, typed as such. No `@types` package, no separate TypeScript build to keep in sync.
|
|
19
35
|
- **Two integration paths, same API.** In-process factory for CLI tools, scripts, single-process services. Optional `thin-gate` subprocess for fault isolation, cross-process quota, and any-language HTTP callers — no code change to switch.
|
|
20
36
|
|
|
21
37
|
## How it compares
|
|
22
38
|
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
39
|
+
**LiteLLM** is the closest analog but lives in Python. **Vercel AI SDK** is an
|
|
40
|
+
application toolkit, not an infra layer. **OpenRouter** is the same one-API
|
|
41
|
+
promise as a SaaS in your request path. **Raw provider SDKs** are N different
|
|
42
|
+
shapes with no cost accounting.
|
|
27
43
|
|
|
28
44
|
| | mohdel | LiteLLM | Vercel AI SDK | OpenRouter | Raw SDKs |
|
|
29
45
|
|---|---|---|---|---|---|
|
|
@@ -32,25 +48,28 @@ Python; **Vercel AI SDK** is an application toolkit, not an infra layer;
|
|
|
32
48
|
| Self-hosted, keys never leave your infra | yes | yes | yes | no | yes |
|
|
33
49
|
| Provider-SDK process isolation | yes (thin-gate) | proxy only | no | n/a | no |
|
|
34
50
|
| OTel spans + metrics out of the box | yes | via callbacks | no | no | no |
|
|
35
|
-
| UI streaming helpers, structured output, agents | no
|
|
51
|
+
| UI streaming helpers, structured output, agents | no | no | yes | no | varies |
|
|
36
52
|
|
|
37
53
|
- **vs LiteLLM** — same core promise (unified calls, cost tracking,
|
|
38
54
|
self-hosted gateway), but Node-native: if your stack is JS, there's no
|
|
39
|
-
Python sidecar to deploy, version, and monitor.
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
55
|
+
Python sidecar to deploy, version, and monitor. LiteLLM's proxy exposes an
|
|
56
|
+
OpenAI-compatible endpoint and admin features (virtual keys, budgets);
|
|
57
|
+
thin-gate speaks its own [wire protocol](PROTOCOL.md), so callers use the JS
|
|
58
|
+
client or implement the protocol. LiteLLM also ships a central price map you
|
|
59
|
+
inherit, where mohdel has you keep your own — your negotiated rates,
|
|
60
|
+
per-model tuning and the tags your code selects on, authored by a coding
|
|
61
|
+
agent from a brief.
|
|
62
|
+
- **vs Vercel AI SDK** — a different layer. The AI SDK is an application
|
|
63
|
+
toolkit (UI streaming, structured outputs, agent loops) with no per-call
|
|
64
|
+
cost, no gateway, no process isolation. It sits above mohdel, which is the
|
|
65
|
+
inference primitive underneath.
|
|
47
66
|
- **vs OpenRouter** — the self-hosted version of the same idea. With a SaaS
|
|
48
67
|
router you accept their uptime, their markup, and your prompts transiting
|
|
49
68
|
their infra. Mohdel goes direct to providers with your keys — and ships an
|
|
50
69
|
`openrouter` adapter for when you want both.
|
|
51
|
-
- **vs raw provider SDKs** —
|
|
52
|
-
|
|
53
|
-
|
|
70
|
+
- **vs raw provider SDKs** — mohdel's envelope is flat and close to the SDKs
|
|
71
|
+
underneath, and `cost` / `tokens` come back normalized, so there are not five
|
|
72
|
+
usage shapes to parse.
|
|
54
73
|
|
|
55
74
|
## Documentation
|
|
56
75
|
|
|
@@ -64,26 +83,73 @@ Python; **Vercel AI SDK** is an application toolkit, not an infra layer;
|
|
|
64
83
|
|
|
65
84
|
## Quick Start
|
|
66
85
|
|
|
67
|
-
|
|
86
|
+
Install, run `mo` to pick a provider and paste your API key, then `mo ask`. Gemini, Groq, Mistral and OpenRouter all have free tiers that need no card, and `mo` lists them first if you have no paid key set.
|
|
87
|
+
|
|
88
|
+
`cost` stays `0` until the catalog carries prices. `mo` pulls the provider's model list, but that list carries ids, not prices. `mo model instructions <provider>` writes a brief carrying the field reference, the provider's own pricing and rate-limit links, and the commands that verify a draft; hand it to the coding agent you already run:
|
|
89
|
+
|
|
90
|
+
```bash
|
|
91
|
+
mo model instructions openai > mohdel-brief.md
|
|
92
|
+
claude "read mohdel-brief.md, then add gpt-5.6-luna to my mohdel catalog"
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
The agent drafts `mohdel-candidate.json` and loops on `mo model check --entry mohdel-candidate.json` until it reports no errors. You run `mo model apply mohdel-candidate.json`, which prints the full diff — including any field the draft would remove — before writing anything. Entries carry `source` and `sourcedAt`, so a price can be traced back to the page it came from.
|
|
96
|
+
|
|
97
|
+
By hand: `mo curate <provider>` with the worked entries in [`config/curated.example.json`](config/curated.example.json), and `mo model set <id> <key> <value>` a field at a time.
|
|
68
98
|
|
|
69
99
|
Model IDs always use the `<provider>/<model>` format:
|
|
70
100
|
|
|
71
101
|
```
|
|
72
|
-
|
|
102
|
+
openai/gpt-5.6-luna
|
|
73
103
|
anthropic/claude-sonnet-4-6
|
|
74
104
|
openai/gpt-5.4-mini
|
|
75
105
|
groq/llama-4-scout-17b-16e-instruct
|
|
76
106
|
```
|
|
77
107
|
|
|
108
|
+
## Attack surface
|
|
109
|
+
|
|
110
|
+
Anthropic's 2026 threat report describes actors compromising AI wrapper
|
|
111
|
+
services built on LiteLLM, using prompt injection to exfiltrate the production
|
|
112
|
+
API keys held in their cloud containers. That attack needs two things: keys
|
|
113
|
+
sitting where a process can read them, and a component that injected content
|
|
114
|
+
can steer into reading them. Mohdel is built so neither is present.
|
|
115
|
+
|
|
116
|
+
- **Nothing executes.** No `eval`, no `new Function`, no `child_process`
|
|
117
|
+
anywhere in the session, factory or library, and no automatic tool loop. A
|
|
118
|
+
prompt-injected response cannot make mohdel read a file, run a shell, or make
|
|
119
|
+
a call of its own. Tool execution belongs to the caller, in the caller's
|
|
120
|
+
process.
|
|
121
|
+
- **No network listener.** `thin-gate` binds **unix sockets**, not TCP, for
|
|
122
|
+
both its data and admin planes, and chmods them `0600` — the default umask
|
|
123
|
+
would otherwise leave them world-connectable. There is no port to reach.
|
|
124
|
+
- **No credential store.** The provider key rides on each call envelope and
|
|
125
|
+
goes straight to the SDK client. Mohdel never accumulates a pool of tenant
|
|
126
|
+
keys, because it never holds one.
|
|
127
|
+
- **The session subprocess starts from an empty environment.** It is given
|
|
128
|
+
back only what the runtime reads — `PATH`, proxy and TLS settings, mohdel's
|
|
129
|
+
own dials, `OTEL_*`. Every `*_API_SK`, cloud credential and database URL the
|
|
130
|
+
host happens to hold is dropped at the process boundary. The session gets
|
|
131
|
+
its key from the envelope, so it has no reason to see any other.
|
|
132
|
+
- **Keys are scrubbed and wiped.** Provider error text has the key removed
|
|
133
|
+
before it reaches `detail`, so a 401 body cannot carry your credential into
|
|
134
|
+
your logs. In Rust, envelope bytes are zeroized after each call.
|
|
135
|
+
|
|
136
|
+
What this does **not** cover: if you run an agent loop, injection can still
|
|
137
|
+
bite there — mohdel moves that risk into your process rather than removing it.
|
|
138
|
+
And `mo` does keep keys on disk in `~/.config/mohdel/environment` (mode
|
|
139
|
+
`0600`), which is a key store, for a developer machine.
|
|
140
|
+
|
|
141
|
+
Report a vulnerability per [SECURITY.md](SECURITY.md).
|
|
142
|
+
|
|
78
143
|
## What mohdel is not
|
|
79
144
|
|
|
80
|
-
|
|
145
|
+
For any of the following, mohdel is the wrong layer. Use it alongside a framework that does them, not instead of one.
|
|
81
146
|
|
|
82
147
|
- **Not an orchestrator.** No chains, no agents, no memory, no prompt templates, no retrieval. Wrap mohdel with LangChain, LangGraph, LlamaIndex, Vercel AI SDK, or your own tool loop — mohdel exposes the inference primitive, orchestration stays in your application.
|
|
83
|
-
- **Not a retry / fallback engine.** Errors are classified (`retryable`, `severity`, `type`)
|
|
84
|
-
- **Not a response cache.** The `cache: true` flag on envelopes is for provider-side prompt caching (Anthropic, OpenAI)
|
|
148
|
+
- **Not a retry / fallback engine.** Errors are classified (`retryable`, `severity`, `type`) for the caller to decide on. Mohdel never retries and never swaps models; the retry budget and the fallback choice are the caller's.
|
|
149
|
+
- **Not a response cache.** The `cache: true` flag on envelopes is for provider-side prompt caching (Anthropic, OpenAI), not mohdel-level memoization of results.
|
|
85
150
|
- **Not a context-window / token manager.** No pre-call token count, no projected-cost guard. The caller owns what goes in the prompt and is the source of truth for what counts.
|
|
86
151
|
- **Not a SaaS proxy.** Self-hosted. Your API keys, your infra. No routing through a third party, no vendor lock-in.
|
|
152
|
+
- **Not an AI wrapper.** `mo model instructions` prints a brief — text. It drives no model, ships no prompts, and spends nothing. The agent that reads it is one you already run, on your own tokens, and it never writes your catalog: `mo model apply` shows you the diff and waits.
|
|
87
153
|
|
|
88
154
|
See [ARCHITECTURE.md §Design principles](ARCHITECTURE.md#design-principles) for the full rationale behind each.
|
|
89
155
|
|
|
@@ -93,7 +159,8 @@ See [ARCHITECTURE.md §Design principles](ARCHITECTURE.md#design-principles) for
|
|
|
93
159
|
# One-shot inference — pipeable
|
|
94
160
|
mo ask anthropic/claude-sonnet-4-6 "explain monads"
|
|
95
161
|
cat article.txt | mo ask openai/gpt-5.4 "summarize in 3 bullets"
|
|
96
|
-
echo "hello" | mo ask
|
|
162
|
+
echo "hello" | mo ask openai/gpt-5.6-luna --json | jq .cost
|
|
163
|
+
mo ask openai/gpt-5.6-luna -q "…" 2>err.log # stderr carries failures only
|
|
97
164
|
|
|
98
165
|
# Streaming
|
|
99
166
|
mo ask anthropic/claude-sonnet-4-6 --stream "write a haiku about recursion"
|
|
@@ -127,7 +194,12 @@ mo setup anthropic # configure API key
|
|
|
127
194
|
mo model add fireworks/deepseek-r1 # add a model manually
|
|
128
195
|
mo model set <model> <key> <value> # set any field on a model
|
|
129
196
|
mo model rm <model> <key> # remove a field
|
|
130
|
-
mo check # validate
|
|
197
|
+
mo check # validate the catalog
|
|
198
|
+
|
|
199
|
+
# Let a coding agent write the entry
|
|
200
|
+
mo model instructions anthropic # brief: fields, doc links, review commands
|
|
201
|
+
mo model check --entry mohdel-candidate.json # validate + diff, no write
|
|
202
|
+
mo model apply mohdel-candidate.json # write, after showing the diff
|
|
131
203
|
|
|
132
204
|
# Rate limits
|
|
133
205
|
mo rl show anthropic # provider or model limits
|
|
@@ -140,6 +212,56 @@ mo bench --tag fast --effort low # suite by tag
|
|
|
140
212
|
|
|
141
213
|
All list/show commands support `--json [fields]` — bare `--json` lists available fields (like `gh`).
|
|
142
214
|
|
|
215
|
+
### Tab completion
|
|
216
|
+
|
|
217
|
+
```bash
|
|
218
|
+
source <(mo completion bash) # add to ~/.bashrc
|
|
219
|
+
```
|
|
220
|
+
|
|
221
|
+
Completes model ids from your catalog, provider names, field names for
|
|
222
|
+
`mo model set`, tags, and the commands themselves — `mo ask gemini/gemini-3.<TAB>`.
|
|
223
|
+
Deprecated ids are left out: they exist so old pins keep resolving, not to be
|
|
224
|
+
picked fresh. Completion reads the catalog directly and never loads the
|
|
225
|
+
inference stack, so a tab press costs about 90ms rather than half a second.
|
|
226
|
+
|
|
227
|
+
### Catalog entries, written by an assistant
|
|
228
|
+
|
|
229
|
+
Provider APIs return model *ids*, not prices — OpenRouter alone publishes
|
|
230
|
+
them, and `mo curate openrouter` fills a catalog unaided. Everything that makes cost
|
|
231
|
+
accounting work — prices, context and output limits, thinking budgets, cache
|
|
232
|
+
rates — is published as prose on a docs page and changes often. `mo model
|
|
233
|
+
instructions` prints a brief that hands your coding agent the field table, the
|
|
234
|
+
provider's reference links, and a verifier it can run in a loop:
|
|
235
|
+
|
|
236
|
+
```bash
|
|
237
|
+
mo model instructions openai > mohdel-brief.md
|
|
238
|
+
```
|
|
239
|
+
|
|
240
|
+
Then start whichever agent you already run on the prompt *read mohdel-brief.md, then
|
|
241
|
+
add gpt-5.6 to my mohdel catalog*:
|
|
242
|
+
|
|
243
|
+
| agent | launch |
|
|
244
|
+
|---|---|
|
|
245
|
+
| Claude Code | `claude "<prompt>"` |
|
|
246
|
+
| Codex CLI | `codex "<prompt>"` |
|
|
247
|
+
| Gemini CLI | `gemini -i "<prompt>"` |
|
|
248
|
+
| opencode | `opencode --prompt "<prompt>"` |
|
|
249
|
+
| Cursor CLI | `cursor-agent "<prompt>"` (installs as `agent` on some platforms) |
|
|
250
|
+
|
|
251
|
+
`mo` asks which one you use and remembers it. It has to be able to fetch a web
|
|
252
|
+
page, because that is where the prices are. Mohdel ships no agent of its own;
|
|
253
|
+
Claude Code, Codex CLI and opencode all install from npm. A session, rather
|
|
254
|
+
than a one-shot, lets you settle which model you want before anything is
|
|
255
|
+
drafted. For a one-shot, pipe instead —
|
|
256
|
+
`mo model instructions openai | claude -p "add gpt-5.6 to my catalog"`, or
|
|
257
|
+
`| codex exec -`.
|
|
258
|
+
|
|
259
|
+
The agent writes `mohdel-candidate.json` and runs `mo model check --entry` until it
|
|
260
|
+
reports no errors; you run `mo model apply`, which prints the full diff —
|
|
261
|
+
including any field the candidate would remove — before writing. Entries carry
|
|
262
|
+
`source` and `sourcedAt` so a price can be traced back to the page it came
|
|
263
|
+
from. See [docs/CATALOG.md](docs/CATALOG.md#editing-with-a-coding-agent).
|
|
264
|
+
|
|
143
265
|
## Library Usage
|
|
144
266
|
|
|
145
267
|
Two integration paths, same adapters underneath: start with the in-process **factory**; graduate to the cross-process **client** when you want gateway-grade isolation.
|
|
@@ -174,7 +296,7 @@ for await (const ev of call(envelope, { socketPath: '/tmp/mohdel-data.sock' }))
|
|
|
174
296
|
}
|
|
175
297
|
```
|
|
176
298
|
|
|
177
|
-
Same API, but inference runs in a pooled subprocess behind the `thin-gate` supervisor (Rust): a crashing provider SDK can't take your service down, quota is enforced across processes, and non-JS callers can speak the same wire. Switching from factory to client is a configuration change, not a rewrite. See [INTEGRATION.md §Client](INTEGRATION.md#
|
|
299
|
+
Same API, but inference runs in a pooled subprocess behind the `thin-gate` supervisor (Rust): a crashing provider SDK can't take your service down, quota is enforced across processes, and non-JS callers can speak the same wire. Switching from factory to client is a configuration change, not a rewrite. See [INTEGRATION.md §Client](INTEGRATION.md#calling-from-javascript) for setup.
|
|
178
300
|
|
|
179
301
|
For the full API — initialization, alias resolution, answer options, response shape, tool use, streaming, vision, error handling, OpenTelemetry, sub-path exports — see **[INTEGRATION.md](INTEGRATION.md)**.
|
|
180
302
|
|
|
@@ -233,7 +355,7 @@ With no session-bin configured, thin-gate runs in demo mode: `POST /v1/call` ret
|
|
|
233
355
|
|
|
234
356
|
### Calling from JS
|
|
235
357
|
|
|
236
|
-
The client snippet under [Library Usage](#library-usage) above is the full surface: `call(envelope, { socketPath, signal? })` returns an async iterable of events. Pass an `AbortSignal` to cancel in flight; thin-gate forwards a cancel control message to the session and reuses it on the pool. The envelope is the flat `answer(prompt, options)` surface plus transport metadata (`callId`, `authId`, `auth.key`, optional `traceparent`); see [`js/core/envelope.js`](js/core/envelope.js) for the full field list.
|
|
358
|
+
The client snippet under [Library Usage](#library-usage) above is the full surface: `call(envelope, { socketPath, signal? })` returns an async iterable of events. Pass an `AbortSignal` to cancel in flight; thin-gate forwards a cancel control message to the session and reuses it on the pool, and the client ends the stream with the same cancelled `done` the in-process path returns (status `incomplete`, warning `cancelled`, the partial output it relayed, zero tokens). The envelope is the flat `answer(prompt, options)` surface plus transport metadata (`callId`, `authId`, `auth.key`, optional `traceparent`); see [`js/core/envelope.js`](js/core/envelope.js) for the full field list.
|
|
237
359
|
|
|
238
360
|
### Other languages
|
|
239
361
|
|
|
@@ -272,7 +394,7 @@ Extending the frozen wire types is breaking — additive changes only on trait m
|
|
|
272
394
|
|
|
273
395
|
### Adding a new provider adapter
|
|
274
396
|
|
|
275
|
-
See [CONTRIBUTING.md](CONTRIBUTING.md#adding-a-session-adapter
|
|
397
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md#adding-a-session-adapter). Short version:
|
|
276
398
|
|
|
277
399
|
1. Create `js/session/adapters/<provider>.js` exporting `async function* <provider>(envelope, { client?, signal? })`.
|
|
278
400
|
2. Map provider-native events to the canonical Event union.
|
|
@@ -298,6 +420,8 @@ FIREWORKS_API_SK=fw_...
|
|
|
298
420
|
DEEPSEEK_API_SK=sk-...
|
|
299
421
|
OPENROUTER_API_SK=sk-or-...
|
|
300
422
|
NOVITA_API_SK=...
|
|
423
|
+
QWEN_API_SK=sk-...
|
|
424
|
+
XIAOMI_API_SK=...
|
|
301
425
|
MOHDEL_LOCAL_API_SK=...
|
|
302
426
|
```
|
|
303
427
|
|
|
@@ -310,8 +434,9 @@ Only set keys for providers you use. Run `mo` with no arguments for interactive
|
|
|
310
434
|
| Path | Purpose |
|
|
311
435
|
|------|---------|
|
|
312
436
|
| `~/.config/mohdel/environment` | API keys |
|
|
313
|
-
| `~/.config/mohdel/default.json` | Default model
|
|
437
|
+
| `~/.config/mohdel/default.json` | Default model, and the coding agent you chose |
|
|
314
438
|
| `~/.config/mohdel/curated.json` | Model catalog with metadata, tags, pricing |
|
|
439
|
+
| `~/.config/mohdel/catalog.local.json` | This installation's own fields and tags, declared for the agent |
|
|
315
440
|
| `~/.config/mohdel/providers.json` | Provider-level rate limits |
|
|
316
441
|
| `~/.config/mohdel/excluded.json` | Excluded models |
|
|
317
442
|
| `~/.cache/mohdel/uploaded-files.json` | Gemini file upload cache |
|
|
@@ -97,14 +97,17 @@
|
|
|
97
97
|
"description": "Human-readable name shown in UIs."
|
|
98
98
|
},
|
|
99
99
|
"description": {
|
|
100
|
-
"type": "string"
|
|
100
|
+
"type": "string",
|
|
101
|
+
"description": "Free-text note about the model, for your own reference."
|
|
101
102
|
},
|
|
102
103
|
"version": {
|
|
103
|
-
"type": "string"
|
|
104
|
+
"type": "string",
|
|
105
|
+
"description": "Provider-side version or snapshot label, when the provider publishes one."
|
|
104
106
|
},
|
|
105
107
|
"createdAt": {
|
|
106
108
|
"type": "string",
|
|
107
|
-
"format": "date-time"
|
|
109
|
+
"format": "date-time",
|
|
110
|
+
"description": "Release timestamp reported by the provider."
|
|
108
111
|
},
|
|
109
112
|
"created": {
|
|
110
113
|
"type": "number",
|
|
@@ -253,6 +256,11 @@
|
|
|
253
256
|
"minimum": 1,
|
|
254
257
|
"description": "Maximum total tokens (input + output)."
|
|
255
258
|
},
|
|
259
|
+
"inputCeilingMargin": {
|
|
260
|
+
"type": "number",
|
|
261
|
+
"minimum": 0,
|
|
262
|
+
"description": "Tokens held back from contextTokenLimit when sizing an input, for a model whose usable window is smaller than the published one. Subtracted by effectiveContextLimit()."
|
|
263
|
+
},
|
|
256
264
|
"outputTokenLimit": {
|
|
257
265
|
"type": "integer",
|
|
258
266
|
"minimum": 1,
|
|
@@ -263,12 +271,8 @@
|
|
|
263
271
|
"minimum": 1,
|
|
264
272
|
"description": "Maximum thinking tokens per call (when separate from output)."
|
|
265
273
|
},
|
|
266
|
-
"tokenizerHeadroom": {
|
|
267
|
-
"type": "number",
|
|
268
|
-
"exclusiveMinimum": 0,
|
|
269
|
-
"description": "Multiplier applied to local token estimates to account for tokenizer drift."
|
|
270
|
-
},
|
|
271
274
|
"thinkingEffortLevels": {
|
|
275
|
+
"description": "Map effort name \u2192 provider-native thinking budget, or null to disable thinking. Standard names: low/medium/high/xhigh/max/none.",
|
|
272
276
|
"oneOf": [
|
|
273
277
|
{
|
|
274
278
|
"type": "null"
|
|
@@ -471,12 +475,30 @@
|
|
|
471
475
|
"description": "[intelligence, speed, latency] triple \u2014 drives 'mo rank'."
|
|
472
476
|
},
|
|
473
477
|
"leaderboardNote": {
|
|
474
|
-
"type": "string"
|
|
478
|
+
"type": "string",
|
|
479
|
+
"description": "Where the 'leaderboard' triple came from."
|
|
475
480
|
},
|
|
476
481
|
"supportsTools": {
|
|
477
482
|
"type": "boolean",
|
|
478
483
|
"description": "Set false to mark a model as tool-less."
|
|
479
484
|
},
|
|
485
|
+
"outputCapStrategy": {
|
|
486
|
+
"type": "string",
|
|
487
|
+
"enum": ["error", "accept"],
|
|
488
|
+
"description": "What this model does when max_tokens exceeds outputTokenLimit: 'error' rejects the call, 'accept' silently serves less. Informational — overrides the provider-level default for embedders building their own requests; mohdel caps the budget either way."
|
|
489
|
+
},
|
|
490
|
+
"reasoningContentPlaceholder": {
|
|
491
|
+
"type": "string",
|
|
492
|
+
"description": "Filler text sent in place of an empty assistant reasoning turn, for OpenAI-compatible providers that reject one. Read by the chat-completions adapter."
|
|
493
|
+
},
|
|
494
|
+
"source": {
|
|
495
|
+
"type": "string",
|
|
496
|
+
"description": "URL the prices and limits in this entry were read from."
|
|
497
|
+
},
|
|
498
|
+
"sourcedAt": {
|
|
499
|
+
"type": "string",
|
|
500
|
+
"description": "Date the 'source' page was last read, as YYYY-MM-DD."
|
|
501
|
+
},
|
|
480
502
|
"imagePrice": {
|
|
481
503
|
"type": "number",
|
|
482
504
|
"minimum": 0,
|
|
@@ -515,7 +537,7 @@
|
|
|
515
537
|
},
|
|
516
538
|
"deprecated": {
|
|
517
539
|
"type": "string",
|
|
518
|
-
"description": "
|
|
540
|
+
"description": "Replacement catalog key. An entry carrying this field is a redirect stub and holds no other fields."
|
|
519
541
|
},
|
|
520
542
|
"suspended": {
|
|
521
543
|
"type": "string",
|
package/js/client/call.js
CHANGED
|
@@ -1,17 +1,42 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Send a CallEnvelope to thin-gate; returns an async iterable of Events.
|
|
3
3
|
*
|
|
4
|
-
* Cancellation: pass an AbortSignal. Aborting
|
|
5
|
-
* thin-gate infers cancel from connection close and
|
|
6
|
-
*
|
|
7
|
-
*
|
|
4
|
+
* Cancellation: pass an AbortSignal. Aborting closes the HTTP request;
|
|
5
|
+
* thin-gate infers cancel from connection close and sends the session a
|
|
6
|
+
* cancel control message. The session's own cancelled terminal is drained
|
|
7
|
+
* gate-side, so the client synthesizes the same cancelled `done` here —
|
|
8
|
+
* partial output from the deltas it relayed, zero tokens, zero cost —
|
|
9
|
+
* giving a caller the same terminal on this path as on the in-process one.
|
|
8
10
|
* @module client/call
|
|
9
11
|
*/
|
|
10
12
|
|
|
11
13
|
import { requestUnix } from './transport.js'
|
|
12
14
|
import { readAll, parseErrorBody } from './response.js'
|
|
13
15
|
import { parseNDJSON } from './ndjson.js'
|
|
14
|
-
import { isEvent, MohdelError } from '#core'
|
|
16
|
+
import { isEvent, MohdelError, STATUS_INCOMPLETE, WARNING_CANCELLED } from '#core'
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* @param {string} start
|
|
20
|
+
* @param {string | null} first
|
|
21
|
+
* @param {string} output
|
|
22
|
+
* @returns {import('#core/events.js').DoneEvent}
|
|
23
|
+
*/
|
|
24
|
+
function cancelledDone (start, first, output) {
|
|
25
|
+
const end = String(process.hrtime.bigint())
|
|
26
|
+
return {
|
|
27
|
+
type: 'done',
|
|
28
|
+
result: {
|
|
29
|
+
status: STATUS_INCOMPLETE,
|
|
30
|
+
output: output || null,
|
|
31
|
+
inputTokens: 0,
|
|
32
|
+
outputTokens: 0,
|
|
33
|
+
thinkingTokens: 0,
|
|
34
|
+
cost: 0,
|
|
35
|
+
timestamps: { start, first: first ?? end, end },
|
|
36
|
+
warning: WARNING_CANCELLED
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
}
|
|
15
40
|
|
|
16
41
|
/**
|
|
17
42
|
* @param {import('#core/envelope.js').CallEnvelope} envelope
|
|
@@ -22,26 +47,58 @@ import { isEvent, MohdelError } from '#core'
|
|
|
22
47
|
* @returns {AsyncGenerator<import('#core/events.js').Event>}
|
|
23
48
|
*/
|
|
24
49
|
export async function * call (envelope, { socketPath, signal, path = '/v1/call' }) {
|
|
25
|
-
const
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
50
|
+
const start = String(process.hrtime.bigint())
|
|
51
|
+
if (signal?.aborted) {
|
|
52
|
+
yield cancelledDone(start, null, '')
|
|
53
|
+
return
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
let res
|
|
57
|
+
try {
|
|
58
|
+
res = await requestUnix({
|
|
59
|
+
socketPath,
|
|
60
|
+
path,
|
|
61
|
+
method: 'POST',
|
|
62
|
+
body: envelope,
|
|
63
|
+
signal
|
|
64
|
+
})
|
|
65
|
+
} catch (e) {
|
|
66
|
+
if (signal?.aborted) {
|
|
67
|
+
yield cancelledDone(start, null, '')
|
|
68
|
+
return
|
|
69
|
+
}
|
|
70
|
+
throw e
|
|
71
|
+
}
|
|
32
72
|
|
|
33
73
|
if (res.statusCode !== 200) {
|
|
34
74
|
const body = await readAll(res)
|
|
35
75
|
throw MohdelError.fromJSON(parseErrorBody(body, res.statusCode ?? 0))
|
|
36
76
|
}
|
|
37
77
|
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
)
|
|
78
|
+
const outputParts = []
|
|
79
|
+
let first = null
|
|
80
|
+
let sawTerminal = false
|
|
81
|
+
try {
|
|
82
|
+
for await (const obj of parseNDJSON(res)) {
|
|
83
|
+
if (!isEvent(obj)) {
|
|
84
|
+
throw new MohdelError(
|
|
85
|
+
'received non-Event object from thin-gate',
|
|
86
|
+
{ type: 'PROTOCOL_INVALID_EVENT', retryable: false }
|
|
87
|
+
)
|
|
88
|
+
}
|
|
89
|
+
if (obj.type === 'delta') {
|
|
90
|
+
if (first === null) first = String(process.hrtime.bigint())
|
|
91
|
+
if (obj.delta?.type === 'message') outputParts.push(obj.delta.delta)
|
|
92
|
+
} else if (obj.type === 'done' || obj.type === 'error') {
|
|
93
|
+
sawTerminal = true
|
|
94
|
+
}
|
|
95
|
+
yield /** @type {import('#core/events.js').Event} */(obj)
|
|
44
96
|
}
|
|
45
|
-
|
|
97
|
+
} catch (e) {
|
|
98
|
+
if (!signal?.aborted) throw e
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
if (signal?.aborted && !sawTerminal) {
|
|
102
|
+
yield cancelledDone(start, first, outputParts.join(''))
|
|
46
103
|
}
|
|
47
104
|
}
|
package/js/client/gate-binary.js
CHANGED
|
@@ -32,6 +32,11 @@
|
|
|
32
32
|
* @throws if no sub-package matches the current host
|
|
33
33
|
*/
|
|
34
34
|
export async function resolveGateBinary () {
|
|
35
|
+
// Advertised by both error messages below, so it has to work: an explicit
|
|
36
|
+
// path wins over the prebuilt package on any platform.
|
|
37
|
+
const override = process.env.MOHDEL_GATE_BINARY
|
|
38
|
+
if (override) return override
|
|
39
|
+
|
|
35
40
|
const pkg = platformPackageName()
|
|
36
41
|
if (!pkg) {
|
|
37
42
|
throw new Error(
|
package/js/client/index.js
CHANGED
package/js/core/envelope.js
CHANGED
|
@@ -35,7 +35,11 @@
|
|
|
35
35
|
* --- Answer options (flat) ---
|
|
36
36
|
*
|
|
37
37
|
* @property {number} [outputBudget]
|
|
38
|
-
* Max output tokens
|
|
38
|
+
* Max output tokens requested. Capped to the spec's `outputTokenLimit`
|
|
39
|
+
* before the provider call — after any thinking headroom the adapter adds,
|
|
40
|
+
* since the sum is what is sent. A spec without that limit cannot be
|
|
41
|
+
* capped and the value goes out as given. See
|
|
42
|
+
* `js/session/adapters/_output_cap.js`.
|
|
39
43
|
* @property {('text'|'json')} [outputType]
|
|
40
44
|
* Default 'text'.
|
|
41
45
|
* @property {('chat'|'coding'|'analysis'|'translation'|'creative')} [outputStyle]
|
package/js/factory/bridge.js
CHANGED
|
@@ -307,8 +307,8 @@ export function configToAuth (configuration) {
|
|
|
307
307
|
* Role mapping:
|
|
308
308
|
* - factory `tool_result` or `tool` → envelope `tool` (carrying
|
|
309
309
|
* `toolCallId`, `content`, and optional `name` from `toolName`).
|
|
310
|
-
*
|
|
311
|
-
*
|
|
310
|
+
* Callers emit the canonical `tool` role directly and the client
|
|
311
|
+
* path preserves it, so the factory path must too.
|
|
312
312
|
* - `assistant.toolCalls` carries through as-is onto the envelope
|
|
313
313
|
* Message so adapters can emit the provider-native tool_use.
|
|
314
314
|
*
|
|
@@ -2,12 +2,6 @@
|
|
|
2
2
|
* Shared `cancelledDone` helper for adapters that need to synthesize
|
|
3
3
|
* a terminal `done` event on `signal.aborted` mid-stream.
|
|
4
4
|
*
|
|
5
|
-
* Three adapters (openai, anthropic, gemini) had byte-identical
|
|
6
|
-
* copies of this; consolidated here. `_chat_completions.js`
|
|
7
|
-
* and `run.js` have their own cancel paths — don't migrate them
|
|
8
|
-
* here unless you're certain the shape matches (thinkingTokens,
|
|
9
|
-
* cost, tool_calls semantics can all differ).
|
|
10
|
-
*
|
|
11
5
|
* @module session/adapters/_cancelled
|
|
12
6
|
*/
|
|
13
7
|
|
|
@@ -17,8 +17,10 @@
|
|
|
17
17
|
*/
|
|
18
18
|
|
|
19
19
|
import { getSpec } from './_catalog.js'
|
|
20
|
+
import { capOutput } from './_output_cap.js'
|
|
20
21
|
import { classifyProviderError } from './_errors.js'
|
|
21
22
|
import { costFor } from './_pricing.js'
|
|
23
|
+
import { cancelledDone } from './_cancelled.js'
|
|
22
24
|
import { catalogKey, bareOf } from '#core/model-id.js'
|
|
23
25
|
import {
|
|
24
26
|
STATUS_COMPLETED,
|
|
@@ -98,6 +100,10 @@ export async function * runChatCompletions (envelope, client, config, deps = {})
|
|
|
98
100
|
try {
|
|
99
101
|
response = await client.chat.completions.create(args, { signal: deps.signal })
|
|
100
102
|
} catch (e) {
|
|
103
|
+
if (deps.signal?.aborted) {
|
|
104
|
+
yield cancelledDone(start, null, envelope, '', 0, 0)
|
|
105
|
+
return
|
|
106
|
+
}
|
|
101
107
|
deps.log?.warn({ err: e }, `[mohdel:${config.provider}] request failed`)
|
|
102
108
|
yield { type: 'error', error: classifyProviderError(e, envelope.auth?.key, { provider: config.provider }) }
|
|
103
109
|
return
|
|
@@ -161,6 +167,10 @@ async function * runStreaming (envelope, client, args, config, start, deps) {
|
|
|
161
167
|
try {
|
|
162
168
|
stream = await client.chat.completions.create(args, { signal: deps.signal })
|
|
163
169
|
} catch (e) {
|
|
170
|
+
if (deps.signal?.aborted) {
|
|
171
|
+
yield cancelledDone(start, null, envelope, '', 0, 0)
|
|
172
|
+
return
|
|
173
|
+
}
|
|
164
174
|
deps.log?.warn({ err: e }, `[mohdel:${config.provider}] request failed`)
|
|
165
175
|
yield { type: 'error', error: classifyProviderError(e, envelope.auth?.key, { provider: config.provider }) }
|
|
166
176
|
return
|
|
@@ -168,6 +178,10 @@ async function * runStreaming (envelope, client, args, config, start, deps) {
|
|
|
168
178
|
|
|
169
179
|
try {
|
|
170
180
|
for await (const chunk of stream) {
|
|
181
|
+
if (deps.signal?.aborted) {
|
|
182
|
+
yield cancelledDone(start, first, envelope, contentParts.join(''), 0, 0)
|
|
183
|
+
return
|
|
184
|
+
}
|
|
171
185
|
const choice = chunk.choices?.[0]
|
|
172
186
|
// DeepSeek V4 / deepseek-reasoner / Cerebras reasoning models emit
|
|
173
187
|
// `delta.reasoning_content` chunks before visible content. Capture
|
|
@@ -225,11 +239,20 @@ async function * runStreaming (envelope, client, args, config, start, deps) {
|
|
|
225
239
|
if (chunk.usage) usage = chunk.usage
|
|
226
240
|
}
|
|
227
241
|
} catch (e) {
|
|
242
|
+
if (deps.signal?.aborted) {
|
|
243
|
+
yield cancelledDone(start, first, envelope, contentParts.join(''), 0, 0)
|
|
244
|
+
return
|
|
245
|
+
}
|
|
228
246
|
deps.log?.warn({ err: e }, `[mohdel:${config.provider}] stream failed`)
|
|
229
247
|
yield { type: 'error', error: classifyProviderError(e, envelope.auth?.key, { provider: config.provider }) }
|
|
230
248
|
return
|
|
231
249
|
}
|
|
232
250
|
|
|
251
|
+
if (deps.signal?.aborted) {
|
|
252
|
+
yield cancelledDone(start, first, envelope, contentParts.join(''), 0, 0)
|
|
253
|
+
return
|
|
254
|
+
}
|
|
255
|
+
|
|
233
256
|
const collectedToolCalls = Object.values(toolCallAccum).map(tc => ({
|
|
234
257
|
id: tc.id,
|
|
235
258
|
function: { name: tc.name, arguments: tc.arguments }
|
|
@@ -370,6 +393,8 @@ function buildRequest (envelope, spec, config) {
|
|
|
370
393
|
args[config.identifierField || 'user'] = envelope.identifier
|
|
371
394
|
}
|
|
372
395
|
|
|
396
|
+
args.max_tokens = capOutput(args.max_tokens, spec?.outputTokenLimit)
|
|
397
|
+
|
|
373
398
|
return args
|
|
374
399
|
}
|
|
375
400
|
|