@aria-framework/ai 0.26.0 → 0.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +47 -13
- package/index.js +24 -18
- package/package.json +4 -3
- package/providers/anthropic.js +28 -4
- package/providers/lmx.js +14 -56
- package/providers/openai-compatible.js +43 -7
- package/reasoning.js +176 -0
package/README.md
CHANGED
|
@@ -35,7 +35,7 @@ const ai = createAiClient({
|
|
|
35
35
|
const r = await ai.complete({
|
|
36
36
|
system, messages, maxTokens: 400,
|
|
37
37
|
schema, // optional: constrained JSON
|
|
38
|
-
// reasoning: 'full', // optional,
|
|
38
|
+
// reasoning: 'full', // optional, every provider: 'off' (default) or 'full' (see Reasoning below)
|
|
39
39
|
});
|
|
40
40
|
// r.text, r.json (when schema), r.model, r.usage.total, r.ms
|
|
41
41
|
```
|
|
@@ -44,6 +44,12 @@ Every failure is thrown as an `AiError` with a `kind`: `disabled`, `unconfigured
|
|
|
44
44
|
`timeout`, `auth`, `rate_limit`, `bad_response` or `refused`. The `message` is written for
|
|
45
45
|
an admin screen and never includes a URL or key.
|
|
46
46
|
|
|
47
|
+
**A failed call that spent tokens is still metered** (0.26.1). When the model answered but the answer
|
|
48
|
+
was unusable (it thought until the limit, or a structured reply was cut off), the error carries
|
|
49
|
+
`usage` (`{ prompt, completion, total }`) and `budget.record(cfg, { usage, model, ms, failed: true }, meta)`
|
|
50
|
+
is called before the error is thrown. A failure that spent nothing (refused, unreachable, timeout)
|
|
51
|
+
records nothing. If `record` throws, the original error still reaches the caller.
|
|
52
|
+
|
|
47
53
|
`providerStore`, `usageStore`, `speedStore` and `lmxStore` are optional db-worker-backed
|
|
48
54
|
stores. Each exports a `schemaFor(dialect)`, so the app's migration can be checked against it.
|
|
49
55
|
|
|
@@ -63,7 +69,7 @@ to write yourself:
|
|
|
63
69
|
| Engine URLs come from the document | Resolved on every call, and never stored |
|
|
64
70
|
| A status outage is not an inference outage | Keeps routing on the last good document for 3 minutes (`staleMs`), logging loudly |
|
|
65
71
|
| Pin the certificate; never disable verification | Uses an undici `Agent({ connect: { ca } })` per stack. Never `NODE_EXTRA_CA_CERTS`, never `rejectUnauthorized:false` |
|
|
66
|
-
| Choose the reasoning flag from the model | By default thinking is kept to a minimum: `qwen` → `chat_template_kwargs.enable_thinking=false`, `gpt-oss` → `reasoning_effort:'low'`, anything else gets neither (with a warning). `reasoning: 'full'` on the call (since 0.26.0) sends `enable_thinking=true` / `reasoning_effort:'high'` instead. The flag always overrides a raw `extra`; only the named option changes it. See [Reasoning](#reasoning-reasoning-full) |
|
|
72
|
+
| Choose the reasoning flag from the model | By default thinking is kept to a minimum: `qwen` → `chat_template_kwargs.enable_thinking=false`, `gpt-oss` → `reasoning_effort:'low'`, anything else gets neither (with a warning). `reasoning: 'full'` on the call (since 0.26.0) sends `enable_thinking=true` / `reasoning_effort:'high'` instead. The flag always overrides a raw `extra`; only the named option changes it. Since 0.27.0 the same rule applies to every provider. See [Reasoning](#reasoning-reasoning-off--full) |
|
|
67
73
|
| Size requests against `maxInputTokens` (per slot) | `contextTokens` is taken from the engine, not from config |
|
|
68
74
|
| Identity is `(instance, id)`, falling back to `name` | `lmx.engineId` is matched first; `rekeyPlan()` migrates rows when ids are minted or engines renamed |
|
|
69
75
|
| A `429` is about the key, not the engine | Fails fast (since 0.25.0): throws `rate_limit` with `lmxSkip: 'lmx_throttled'` and `retryAfterMs`. The client never waits; whether to wait or move on is the caller's decision |
|
|
@@ -170,10 +176,11 @@ instead:
|
|
|
170
176
|
Failover across several engines for one job is the app's job; lmx provides none. List
|
|
171
177
|
endpoints in preference order and take the first one that serves.
|
|
172
178
|
|
|
173
|
-
### Reasoning (`reasoning: 'full'`)
|
|
179
|
+
### Reasoning (`reasoning: 'off' | 'full'`)
|
|
174
180
|
|
|
175
|
-
|
|
176
|
-
|
|
181
|
+
How much a model may think is decided by the **app, per call**, and every provider honours it (since
|
|
182
|
+
0.27.0; before that only lmx did). Leave it out, or pass `reasoning: 'off'`, and the model thinks as
|
|
183
|
+
little as its family allows, so a quick answer stays quick. Pass `reasoning: 'full'` to let it think:
|
|
177
184
|
|
|
178
185
|
```js
|
|
179
186
|
const r = await ai.complete({
|
|
@@ -198,23 +205,50 @@ a `bad_response` AiError saying it ran out of room (and, when the server says so
|
|
|
198
205
|
tokens went on internal reasoning), but the work is lost either way. Use **`maxTokens` ≥ 4096** and a
|
|
199
206
|
**`timeoutMs` ≥ 120 s** (120000) on the endpoint, or the deadline cuts the answer off first.
|
|
200
207
|
|
|
201
|
-
**
|
|
208
|
+
**lmx, openai-compatible and lmstudio: what is added to the request body, per model family.** The
|
|
209
|
+
family is read from the model the engine is running (lmx) or the endpoint's model name (others):
|
|
202
210
|
|
|
203
|
-
| Family |
|
|
211
|
+
| Family | `'off'` (the default) | `reasoning: 'full'` |
|
|
204
212
|
|---|---|---|
|
|
205
213
|
| qwen | `{"chat_template_kwargs":{"enable_thinking":false}}` | `{"chat_template_kwargs":{"enable_thinking":true}}` |
|
|
206
214
|
| gpt-oss | `{"reasoning_effort":"low"}` | `{"reasoning_effort":"high"}` |
|
|
207
|
-
| anything else | nothing (warning
|
|
215
|
+
| anything else | nothing (lmx logs a warning) | nothing (a warning that `'full'` could not be honoured) |
|
|
216
|
+
|
|
217
|
+
**LM Studio ignores `chat_template_kwargs`** (measured live: Qwen3 thought just as long with
|
|
218
|
+
`enable_thinking: false`). So on **`provider: 'lmstudio'`** a qwen model gets two more switches. One is
|
|
219
|
+
`{"reasoning_effort":"none"}` for `'off'` and `{"reasoning_effort":"high"}` for `'full'`: verified live, but
|
|
220
|
+
LM Studio documents `reasoning_effort` only for gpt-oss. The other is Qwen's own documented prompt switch,
|
|
221
|
+
`/no_think` for `'off'` and `/think` for `'full'`, appended to the system prompt (or sent as one when there
|
|
222
|
+
is none): documented on LM Studio's Qwen3 model pages, but only hybrid Qwen3 reads it. Both are sent, so
|
|
223
|
+
if a future LM Studio drops one the other still holds. Configure an LM Studio server as `lmstudio`, not
|
|
224
|
+
`openai-compatible`; as `openai-compatible` it keeps thinking on every call. Plain `openai-compatible`
|
|
225
|
+
gets neither, because vLLM validates `reasoning_effort` and may refuse `'none'`.
|
|
226
|
+
|
|
227
|
+
**anthropic: what is sent, per Claude generation.** The Messages API differs by generation: Opus 5.5
|
|
228
|
+
and Fable cannot turn thinking off (effort is the only lever), Sonnet 5.5 turns it off with
|
|
229
|
+
`between_tools`, and several current models refuse `temperature`, so none is sent to them.
|
|
230
|
+
|
|
231
|
+
| Models | `'off'` (the default) | `reasoning: 'full'` | temperature |
|
|
232
|
+
|---|---|---|---|
|
|
233
|
+
| claude-fable / claude-opus-5-5 | `{"output_config":{"effort":"low"}}` | `{"output_config":{"effort":"high"}}` | not sent |
|
|
234
|
+
| claude-opus-5 | `{"output_config":{"effort":"low"}}` | `{"thinking":{"type":"adaptive"},"output_config":{"effort":"high"}}` | not sent |
|
|
235
|
+
| claude-sonnet-5-5 | `{"thinking":{"type":"between_tools"}}` | `{"thinking":{"type":"adaptive"},"output_config":{"effort":"high"}}` | not sent |
|
|
236
|
+
| claude-sonnet-5 | `{"thinking":{"type":"disabled"}}` | `{"thinking":{"type":"adaptive"},"output_config":{"effort":"high"}}` | not sent |
|
|
237
|
+
| claude-opus-4-7 / 4-8 | nothing | `{"thinking":{"type":"adaptive"},"output_config":{"effort":"high"}}` | not sent |
|
|
238
|
+
| claude-*-4-6 | nothing | `{"thinking":{"type":"adaptive"},"output_config":{"effort":"high"}}` | sent |
|
|
239
|
+
| claude-*-4-5 and older | nothing | thinking enabled with a budget of `maxTokens` (at least 1024), added on top of `max_tokens` | sent; not sent with `'full'` |
|
|
240
|
+
| anything else | nothing | nothing (a warning that `'full'` could not be honoured) | sent |
|
|
241
|
+
|
|
242
|
+
`effort` is merged into any `output_config` the structured-output schema already put there.
|
|
208
243
|
|
|
209
244
|
Rules:
|
|
210
245
|
|
|
211
|
-
- `'
|
|
212
|
-
for example `'high'`, `true` or `''`, is refused with `AiError` kind `refused` before any
|
|
246
|
+
- The values are `'off'` and `'full'`. Leaving the option out (or `null`) means `'off'`. Any other
|
|
247
|
+
value, for example `'high'`, `true` or `''`, is refused with `AiError` kind `refused` before any
|
|
213
248
|
request is made, whichever provider the endpoint uses.
|
|
214
249
|
- The flag is merged **after** `extra`, so a raw `extra: { chat_template_kwargs: … }` or
|
|
215
|
-
`extra: { reasoning_effort: … }` cannot change it. Only `reasoning` can.
|
|
216
|
-
|
|
217
|
-
warning, removes the option, and nothing reaches that provider's request body.
|
|
250
|
+
`extra: { reasoning_effort: … }` cannot change it. Only `reasoning` can. The option itself is never
|
|
251
|
+
copied into a request body.
|
|
218
252
|
- `REASONING_MODES` (package root) lists the accepted values for an app that validates its own
|
|
219
253
|
settings.
|
|
220
254
|
|
package/index.js
CHANGED
|
@@ -26,7 +26,7 @@ const facts = require('./facts');
|
|
|
26
26
|
const { AiError, fromFetchFailure, redact } = require('./error');
|
|
27
27
|
const { polish } = require('./polish');
|
|
28
28
|
const { generate } = require('./generate');
|
|
29
|
-
const { assertReasoning, REASONING_MODES } = require('./
|
|
29
|
+
const { assertReasoning, REASONING_MODES } = require('./reasoning');
|
|
30
30
|
|
|
31
31
|
const PROVIDERS = {
|
|
32
32
|
// 'lmstudio' and 'openai-compatible' are the SAME adapter with different defaults — a kindness to
|
|
@@ -52,9 +52,6 @@ const DEFAULTS = {
|
|
|
52
52
|
const RETRY_AFTER_MS = 400;
|
|
53
53
|
const RETRY_ONLY_IF_FAILED_WITHIN_MS = 5000;
|
|
54
54
|
|
|
55
|
-
/** Providers already warned that they ignore `reasoning` — once per provider per process. */
|
|
56
|
-
const warnedReasoning = new Set();
|
|
57
|
-
|
|
58
55
|
const NOOP_LOGGER = { info() {}, warn() {}, error() {} };
|
|
59
56
|
const NOOP_BUDGET = { async assertWithinBudget() {}, async record() {} };
|
|
60
57
|
|
|
@@ -106,7 +103,7 @@ function createAiClient(deps = {}) {
|
|
|
106
103
|
* Ask the configured model for something.
|
|
107
104
|
* @param {{system?:string, messages:Array, maxTokens?:number, temperature?:number,
|
|
108
105
|
* schema?:object, signal?:AbortSignal, ticketId?:*, skipBudget?:boolean,
|
|
109
|
-
* reasoning?:'full'}} opts
|
|
106
|
+
* reasoning?:'off'|'full'}} opts
|
|
110
107
|
* @param {object} [cfgOverride] the resolved config, when the caller already has it
|
|
111
108
|
*/
|
|
112
109
|
async function complete(opts, cfgOverride) {
|
|
@@ -134,22 +131,31 @@ function createAiClient(deps = {}) {
|
|
|
134
131
|
// by whichever caller forgets. `skipBudget` is for the admin's Test connection; it still records.
|
|
135
132
|
if (!opts.skipBudget) await meter.assertWithinBudget(cfg, { ticketId: opts.ticketId });
|
|
136
133
|
|
|
137
|
-
//
|
|
138
|
-
//
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
134
|
+
// EVERY PROVIDER HONOURS `reasoning` (0.27.0): each adapter translates it for its model family
|
|
135
|
+
// (see reasoning.js). It used to be lmx-only, warned about and stripped everywhere else.
|
|
136
|
+
const callOpts = opts;
|
|
137
|
+
|
|
138
|
+
const started = Date.now();
|
|
139
|
+
let result;
|
|
140
|
+
try {
|
|
141
|
+
result = await withOneRetry(() => adapter.complete(cfg, callOpts), opts);
|
|
142
|
+
} catch (err) {
|
|
143
|
+
// A FAILED CALL CAN STILL HAVE SPENT TOKENS (0.26.1) — a model that thought until it ran out of
|
|
144
|
+
// room, a structured reply cut off mid-bracket. Recording only successes let a benchmark or a
|
|
145
|
+
// connection test on such a model run outside the ceiling. Adapters put normalised usage on
|
|
146
|
+
// the error; a failure that spent nothing (refused, unreachable) carries none and is not counted.
|
|
147
|
+
// The ledger failing must never replace the error the caller needs to see.
|
|
148
|
+
if (err && err.usage && err.usage.total > 0) {
|
|
149
|
+
try {
|
|
150
|
+
await meter.record(cfg, { usage: err.usage, model: cfg.model, ms: Date.now() - started, failed: true },
|
|
151
|
+
{ ticketId: opts.ticketId });
|
|
152
|
+
} catch (recErr) {
|
|
153
|
+
log.warn(`AI: usage of a failed call not recorded (${recErr.message})`);
|
|
154
|
+
}
|
|
146
155
|
}
|
|
147
|
-
|
|
148
|
-
callOpts = rest;
|
|
156
|
+
throw err;
|
|
149
157
|
}
|
|
150
158
|
|
|
151
|
-
const result = await withOneRetry(() => adapter.complete(cfg, callOpts), opts);
|
|
152
|
-
|
|
153
159
|
// ...and the counter after it, AWAITED: two calls in quick succession must both be counted
|
|
154
160
|
// before the second's ceiling check reads the total, or the limit is enforced against a stale one.
|
|
155
161
|
await meter.record(cfg, result, { ticketId: opts.ticketId });
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@aria-framework/ai",
|
|
3
|
-
"description": "Aria App Framework
|
|
4
|
-
"version": "0.
|
|
3
|
+
"description": "Aria App Framework — AI module. A dependency-injected model seam (createAiClient) over several providers (LM Studio / OpenAI-compatible / Anthropic), with a fact-preservation guard, generic Polish and Generate writing engines, and a browser polish widget. Prompts and config stay in the consuming app.",
|
|
4
|
+
"version": "0.27.0",
|
|
5
5
|
"license": "UNLICENSED",
|
|
6
6
|
"private": false,
|
|
7
7
|
"publishConfig": {
|
|
@@ -29,6 +29,7 @@
|
|
|
29
29
|
"providers/lmxDiscovery.js",
|
|
30
30
|
"providers/lmxTransport.js",
|
|
31
31
|
"providers/openai-compatible.js",
|
|
32
|
+
"reasoning.js",
|
|
32
33
|
"speedStore.js",
|
|
33
34
|
"untrusted.js",
|
|
34
35
|
"usageStore.js",
|
|
@@ -47,7 +48,7 @@
|
|
|
47
48
|
}
|
|
48
49
|
},
|
|
49
50
|
"scripts": {
|
|
50
|
-
"test": "node test/smoke.js && node test/usageStore.js && node test/providerStore.js && node test/speedStore.js && node test/health.js && node test/listModels.js && node test/benchmark.js && node test/lmxDiscovery.js && node test/lmx.js && node test/lmxVerify.js && node test/lmxStore.js && node test/lmxStatus.js && node test/jobCard.js && node test/untrusted.js && node test/packaging.js && node test/views.js && node test/fenced.js && node test/polish.js && node test/reasoning.js",
|
|
51
|
+
"test": "node test/smoke.js && node test/usageStore.js && node test/providerStore.js && node test/speedStore.js && node test/health.js && node test/listModels.js && node test/benchmark.js && node test/lmxDiscovery.js && node test/lmx.js && node test/lmxVerify.js && node test/lmxStore.js && node test/lmxStatus.js && node test/jobCard.js && node test/untrusted.js && node test/packaging.js && node test/views.js && node test/fenced.js && node test/polish.js && node test/reasoning.js && node test/reasoningProviders.js && node test/usageOnFailure.js",
|
|
51
52
|
"prepublishOnly": "node ../../test/packaging.js ai"
|
|
52
53
|
},
|
|
53
54
|
"devDependencies": {
|
package/providers/anthropic.js
CHANGED
|
@@ -19,6 +19,7 @@
|
|
|
19
19
|
'use strict';
|
|
20
20
|
|
|
21
21
|
const { AiError, fromFetchFailure, redact } = require('../error');
|
|
22
|
+
const { anthropicReasoning } = require('../reasoning');
|
|
22
23
|
|
|
23
24
|
const API_VERSION = '2023-06-01';
|
|
24
25
|
const DEFAULT_BASE = 'https://api.anthropic.com/v1';
|
|
@@ -60,6 +61,22 @@ async function complete(cfg, opts) {
|
|
|
60
61
|
if (beta) headers['anthropic-beta'] = beta;
|
|
61
62
|
}
|
|
62
63
|
|
|
64
|
+
// HOW MUCH CLAUDE MAY THINK (0.27.0) — per model generation, because the API differs by
|
|
65
|
+
// generation (see ../reasoning.js). Absent means 'off'. An unknown model keeps the pre-0.27 body.
|
|
66
|
+
const r = anthropicReasoning(cfg.model, opts.reasoning, body.max_tokens);
|
|
67
|
+
if (r) {
|
|
68
|
+
if (r.thinking) body.thinking = r.thinking;
|
|
69
|
+
// Merged into output_config, which may already carry the structured-output format.
|
|
70
|
+
if (r.effort) body.output_config = { ...(body.output_config || {}), effort: r.effort };
|
|
71
|
+
// Several current models reject temperature outright, and extended thinking on older ones
|
|
72
|
+
// requires the default. Sending it would be a 400, so it is left out.
|
|
73
|
+
if (!r.sendTemperature) delete body.temperature;
|
|
74
|
+
body.max_tokens = r.maxTokens;
|
|
75
|
+
} else if (opts.reasoning === 'full' && cfg.logger) {
|
|
76
|
+
cfg.logger.warn(`${label}: reasoning 'full' was asked for, but model "${cfg.model}" is not in the `
|
|
77
|
+
+ 'known Claude generations — sending no thinking setting; the model runs on its own default.');
|
|
78
|
+
}
|
|
79
|
+
|
|
63
80
|
const started = Date.now();
|
|
64
81
|
const controller = new AbortController();
|
|
65
82
|
const timer = setTimeout(() => controller.abort(), cfg.timeoutMs || 60000);
|
|
@@ -97,6 +114,13 @@ async function complete(cfg, opts) {
|
|
|
97
114
|
clearTimeout(timer);
|
|
98
115
|
}
|
|
99
116
|
|
|
117
|
+
// A 200 WHOSE BODY IS `null` (0.26.1) is a typed failure, not a TypeError further down.
|
|
118
|
+
if (!payload || typeof payload !== 'object') {
|
|
119
|
+
throw new AiError('bad_response', `${label} answered with an empty body.`);
|
|
120
|
+
}
|
|
121
|
+
// Spent tokens ride on every later error so the client can meter a failed call (0.26.1).
|
|
122
|
+
const usage = normaliseUsage(payload.usage);
|
|
123
|
+
|
|
100
124
|
// Difference 4: blocks, not a single string. Concatenated so a model that splits its answer over
|
|
101
125
|
// two text blocks does not silently lose the second one.
|
|
102
126
|
// Thinking arrives as its own block TYPE here rather than a field, so filtering to 'text' already
|
|
@@ -109,7 +133,7 @@ async function complete(cfg, opts) {
|
|
|
109
133
|
const thought = (payload.content || []).some((b) => b && b.type === 'thinking');
|
|
110
134
|
throw new AiError('bad_response', thought
|
|
111
135
|
? `${label} returned only its internal reasoning. Raise the reply limit and try again.`
|
|
112
|
-
: `${label} returned no text (stop reason: ${payload.stop_reason || 'none given'})
|
|
136
|
+
: `${label} returned no text (stop reason: ${payload.stop_reason || 'none given'}).`, { usage });
|
|
113
137
|
}
|
|
114
138
|
|
|
115
139
|
let json = null;
|
|
@@ -121,10 +145,10 @@ async function complete(cfg, opts) {
|
|
|
121
145
|
if (payload.stop_reason === 'max_tokens') {
|
|
122
146
|
throw new AiError('bad_response',
|
|
123
147
|
`${label} ran out of room mid-answer — the structured reply was cut off before it was ` +
|
|
124
|
-
'finished. A Regenerate usually succeeds.', { cause: err });
|
|
148
|
+
'finished. A Regenerate usually succeeds.', { cause: err, usage });
|
|
125
149
|
}
|
|
126
150
|
throw new AiError('bad_response',
|
|
127
|
-
`${label} was asked for structured output and returned text that will not parse.`, { cause: err });
|
|
151
|
+
`${label} was asked for structured output and returned text that will not parse.`, { cause: err, usage });
|
|
128
152
|
}
|
|
129
153
|
}
|
|
130
154
|
|
|
@@ -132,7 +156,7 @@ async function complete(cfg, opts) {
|
|
|
132
156
|
text,
|
|
133
157
|
json,
|
|
134
158
|
model: payload.model || cfg.model,
|
|
135
|
-
usage
|
|
159
|
+
usage,
|
|
136
160
|
ms: Date.now() - started,
|
|
137
161
|
finishReason: payload.stop_reason || null,
|
|
138
162
|
truncated: payload.stop_reason === 'max_tokens',
|
package/providers/lmx.js
CHANGED
|
@@ -137,56 +137,15 @@ function skipError(name, res) {
|
|
|
137
137
|
}
|
|
138
138
|
|
|
139
139
|
/**
|
|
140
|
-
* THE REASONING FLAG IS A PROPERTY OF THE MODEL, NOT OF THE ENGINE
|
|
141
|
-
*
|
|
142
|
-
* Both families reason before answering, and each needs a different instruction to stop. Getting it
|
|
143
|
-
* wrong costs the whole budget and returns nothing — `content: ""` with `finish_reason: "length"`,
|
|
144
|
-
* no error, and a retry produces the same nothing.
|
|
140
|
+
* THE REASONING FLAG IS A PROPERTY OF THE MODEL, NOT OF THE ENGINE.
|
|
145
141
|
*
|
|
146
142
|
* Read from the live document every call, never stored: point an engine at a different model and
|
|
147
|
-
* the correct flag changes underneath a configuration that never changed.
|
|
148
|
-
*
|
|
149
|
-
*
|
|
150
|
-
*
|
|
151
|
-
* exists to avoid.
|
|
152
|
-
*
|
|
153
|
-
* THE DEFAULT IS "THINK AS LITTLE AS THE FAMILY ALLOWS"; `reasoning: 'full'` (0.26.0) is the one
|
|
154
|
-
* named way to ask for the opposite — Qwen thinking on, gpt-oss effort high — for judgement work
|
|
155
|
-
* (triage) where a thinking-off answer measured as a different product. It is a NAMED per-call
|
|
156
|
-
* option rather than a raw `extra`, because the forced flag deliberately outranks `extra`: a stray
|
|
157
|
-
* passthrough must never be able to turn thinking on and spend a budget sized for a quick answer.
|
|
158
|
-
*/
|
|
159
|
-
const REASONING_MODES = Object.freeze(['full']);
|
|
160
|
-
|
|
161
|
-
/**
|
|
162
|
-
* Refuse a `reasoning` value this package does not define. `undefined`/`null` mean the default.
|
|
163
|
-
* Anything else — 'high', 'Full', true, '' — is a caller's mistake, and silently treating it as the
|
|
164
|
-
* default would hand a triage caller a thinking-off answer while it believed it had asked for more.
|
|
165
|
-
*/
|
|
166
|
-
function assertReasoning(value) {
|
|
167
|
-
if (value === undefined || value === null) return;
|
|
168
|
-
if (!REASONING_MODES.includes(value)) {
|
|
169
|
-
throw new AiError('refused',
|
|
170
|
-
`Unknown reasoning option ${JSON.stringify(value)}. The only value is `
|
|
171
|
-
+ `${REASONING_MODES.map((m) => `'${m}'`).join(', ')}; leave it out for the default.`);
|
|
172
|
-
}
|
|
173
|
-
}
|
|
174
|
-
|
|
175
|
-
/**
|
|
176
|
-
* The families with a known flag, in match order. A TABLE rather than a chain of ifs so the README
|
|
177
|
-
* test can enumerate it: a family added here and not documented fails that test.
|
|
143
|
+
* the correct flag changes underneath a configuration that never changed. The table and the modes
|
|
144
|
+
* moved to ../reasoning.js in 0.27.0, when every provider started honouring `reasoning`; they are
|
|
145
|
+
* re-exported below so existing imports keep working. openai-compatible applies the flag, from the
|
|
146
|
+
* engine's model (`reasoningModel` on the engine config) — one place decides what is sent.
|
|
178
147
|
*/
|
|
179
|
-
const REASONING_FAMILIES =
|
|
180
|
-
{ family: 'qwen', match: /qwen/, flag: (full) => ({ chat_template_kwargs: { enable_thinking: full } }) },
|
|
181
|
-
{ family: 'gpt-oss', match: /gpt-oss/, flag: (full) => ({ reasoning_effort: full ? 'high' : 'low' }) }
|
|
182
|
-
]);
|
|
183
|
-
|
|
184
|
-
function reasoningFor(modelPath, mode) {
|
|
185
|
-
assertReasoning(mode);
|
|
186
|
-
const m = String(modelPath || '').toLowerCase();
|
|
187
|
-
const f = REASONING_FAMILIES.find((x) => x.match.test(m));
|
|
188
|
-
return f ? f.flag(mode === 'full') : null;
|
|
189
|
-
}
|
|
148
|
+
const { REASONING_MODES, REASONING_FAMILIES, assertReasoning, reasoningFor } = require('../reasoning');
|
|
190
149
|
|
|
191
150
|
/** The config openai-compatible needs, once the address is known. */
|
|
192
151
|
function engineConfig(cfg, url, engine) {
|
|
@@ -204,7 +163,10 @@ function engineConfig(cfg, url, engine) {
|
|
|
204
163
|
: lmxTransport(cfg.lmx && cfg.lmx.ca),
|
|
205
164
|
// SIZE AGAINST WHAT THE ENGINE REPORTS, not what somebody typed. maxInputTokens is PER SLOT,
|
|
206
165
|
// because -c in llama.cpp is a pool divided across slots.
|
|
207
|
-
contextTokens: (engine && engine.maxInputTokens) || cfg.contextTokens
|
|
166
|
+
contextTokens: (engine && engine.maxInputTokens) || cfg.contextTokens,
|
|
167
|
+
// THE MODEL THE FLAG IS CHOSEN FROM — the one the engine is running, not the endpoint's model
|
|
168
|
+
// name (an lmx endpoint stores none). openai-compatible reads this before cfg.model.
|
|
169
|
+
reasoningModel: engine && engine.model
|
|
208
170
|
};
|
|
209
171
|
}
|
|
210
172
|
|
|
@@ -254,14 +216,10 @@ async function complete(cfg, opts) {
|
|
|
254
216
|
);
|
|
255
217
|
}
|
|
256
218
|
|
|
257
|
-
// The
|
|
258
|
-
|
|
259
|
-
return openai.complete(engineConfig(cfg, res.url, res.engine), {
|
|
260
|
-
|
|
261
|
-
// Merged rather than replacing: a caller's own extras survive. The flag is spread LAST, so a
|
|
262
|
-
// raw extra can never override it — only the named `reasoning` option changes what it says.
|
|
263
|
-
extra: { ...(opts && opts.extra), ...(reasoning || {}) }
|
|
264
|
-
}).catch((err) => { throw tagThrottle(err); });
|
|
219
|
+
// The mode goes on to openai-compatible, which merges the flag AFTER `extra` (0.27.0) — so a raw
|
|
220
|
+
// extra still cannot override it, and the option itself never reaches the request body.
|
|
221
|
+
return openai.complete(engineConfig(cfg, res.url, res.engine), opts || {})
|
|
222
|
+
.catch((err) => { throw tagThrottle(err); });
|
|
265
223
|
}
|
|
266
224
|
|
|
267
225
|
async function embed(cfg, texts) {
|
|
@@ -12,6 +12,7 @@
|
|
|
12
12
|
|
|
13
13
|
'use strict';
|
|
14
14
|
|
|
15
|
+
const { reasoningFor, lmstudioPromptSwitch } = require('../reasoning');
|
|
15
16
|
const { AiError, fromFetchFailure, redact } = require('../error');
|
|
16
17
|
|
|
17
18
|
/**
|
|
@@ -65,6 +66,30 @@ async function complete(cfg, opts) {
|
|
|
65
66
|
};
|
|
66
67
|
}
|
|
67
68
|
|
|
69
|
+
// HOW MUCH THE MODEL MAY THINK (0.27.0), from the model family — the engine's model on lmx
|
|
70
|
+
// (`reasoningModel`), the configured name otherwise. Merged AFTER `extra`, so a raw passthrough
|
|
71
|
+
// cannot override it; only the named `reasoning` option changes it, and that option is never
|
|
72
|
+
// copied into the body itself. Absent means 'off': before 0.27.0 an LM Studio endpoint running
|
|
73
|
+
// Qwen3 reasoned on every call by default, spending budgets sized for a quick answer.
|
|
74
|
+
const flag = reasoningFor(cfg.reasoningModel || cfg.model, opts.reasoning, cfg.provider);
|
|
75
|
+
if (flag) Object.assign(body, flag);
|
|
76
|
+
else if (opts.reasoning === 'full' && cfg.logger && !cfg.lmx) {
|
|
77
|
+
// lmx warns for itself, naming the engine; this is the plain-endpoint case.
|
|
78
|
+
cfg.logger.warn(`${label}: reasoning 'full' was asked for, but no reasoning flag is known for model `
|
|
79
|
+
+ `"${cfg.model}" — sending none; the model runs on its own default.`);
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
// LM STUDIO ALSO GETS QWEN'S DOCUMENTED PROMPT SWITCH (see reasoning.js LMSTUDIO_PROMPT_SWITCH),
|
|
83
|
+
// on the system prompt — appended to the caller's own, or as one of its own when there is none.
|
|
84
|
+
const promptSwitch = lmstudioPromptSwitch(cfg.reasoningModel || cfg.model, opts.reasoning, cfg.provider);
|
|
85
|
+
if (promptSwitch) {
|
|
86
|
+
if (messages.length && messages[0].role === 'system' && typeof messages[0].content === 'string') {
|
|
87
|
+
messages[0] = { role: 'system', content: `${messages[0].content}\n\n${promptSwitch}` };
|
|
88
|
+
} else {
|
|
89
|
+
messages.unshift({ role: 'system', content: promptSwitch });
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
|
|
68
93
|
const started = Date.now();
|
|
69
94
|
const controller = new AbortController();
|
|
70
95
|
const timer = setTimeout(() => controller.abort(), cfg.timeoutMs || 60000);
|
|
@@ -133,7 +158,13 @@ async function complete(cfg, opts) {
|
|
|
133
158
|
// completion" — which sent this investigation after a thinking-model theory that was not the
|
|
134
159
|
// cause. Verified against the real server: the same request on /v1/chat/completions answers
|
|
135
160
|
// normally.
|
|
136
|
-
|
|
161
|
+
// A 200 WHOSE BODY IS `null` (0.26.1) — valid JSON, no object. It used to reach `payload.usage`
|
|
162
|
+
// below as a TypeError, which a router reads as an unknown fault.
|
|
163
|
+
if (!payload || typeof payload !== 'object') {
|
|
164
|
+
throw new AiError('bad_response', `${label} answered with an empty body.`);
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
if (payload.error && !payload.choices) {
|
|
137
168
|
const said = typeof payload.error === 'string'
|
|
138
169
|
? payload.error
|
|
139
170
|
: (payload.error.message || JSON.stringify(payload.error));
|
|
@@ -170,6 +201,10 @@ async function complete(cfg, opts) {
|
|
|
170
201
|
throw emptyCompletion({ label, finishReason, reasoning, maxTokens: body.max_tokens, usage: payload.usage });
|
|
171
202
|
}
|
|
172
203
|
|
|
204
|
+
// TOKENS WERE SPENT EVEN WHEN THE ANSWER IS UNUSABLE (0.26.1). Carried on every error raised after
|
|
205
|
+
// this point, normalised, so the client can meter a failed call — see index.js complete().
|
|
206
|
+
const usage = normaliseUsage(payload.usage);
|
|
207
|
+
|
|
173
208
|
// With a schema the content is still a STRING containing JSON — the server constrains the shape,
|
|
174
209
|
// it does not parse for you. Parsing here rather than at each call site means a malformed body is
|
|
175
210
|
// one error type instead of a surprise in a page render.
|
|
@@ -188,10 +223,10 @@ async function complete(cfg, opts) {
|
|
|
188
223
|
throw new AiError('bad_response',
|
|
189
224
|
`${label} ran out of room mid-answer (${body.max_tokens} tokens) — the structured reply ` +
|
|
190
225
|
'was cut off before it was finished. If this repeats, the schema may be letting the ' +
|
|
191
|
-
'model ramble; a Regenerate usually succeeds.', { cause: err });
|
|
226
|
+
'model ramble; a Regenerate usually succeeds.', { cause: err, usage });
|
|
192
227
|
}
|
|
193
228
|
throw new AiError('bad_response',
|
|
194
|
-
`${label} was asked for structured output and returned text that will not parse.`, { cause: err });
|
|
229
|
+
`${label} was asked for structured output and returned text that will not parse.`, { cause: err, usage });
|
|
195
230
|
}
|
|
196
231
|
}
|
|
197
232
|
|
|
@@ -199,7 +234,7 @@ async function complete(cfg, opts) {
|
|
|
199
234
|
text,
|
|
200
235
|
json,
|
|
201
236
|
model: payload.model || cfg.model,
|
|
202
|
-
usage
|
|
237
|
+
usage,
|
|
203
238
|
ms: Date.now() - started,
|
|
204
239
|
finishReason,
|
|
205
240
|
// A caller that cares (the summary's coverage line) can tell a complete answer from a cut-off
|
|
@@ -236,8 +271,9 @@ function unclosedThinking(text) {
|
|
|
236
271
|
* payload, so they are distinguished: the budget ran out, the budget went on thinking, or the server
|
|
237
272
|
* genuinely said nothing.
|
|
238
273
|
*/
|
|
239
|
-
function emptyCompletion({ label, finishReason, reasoning, maxTokens, usage }) {
|
|
240
|
-
const spent = (
|
|
274
|
+
function emptyCompletion({ label, finishReason, reasoning, maxTokens, usage: raw }) {
|
|
275
|
+
const spent = (raw && raw.completion_tokens) || 0;
|
|
276
|
+
const usage = normaliseUsage(raw);
|
|
241
277
|
if (finishReason === 'length' || (reasoning && !spentLeftRoom(spent, maxTokens))) {
|
|
242
278
|
return new AiError('bad_response',
|
|
243
279
|
`${label} ran out of room before it answered — it used all ${maxTokens} reply tokens` +
|
|
@@ -253,7 +289,7 @@ function emptyCompletion({ label, finishReason, reasoning, maxTokens, usage }) {
|
|
|
253
289
|
}
|
|
254
290
|
return new AiError('bad_response',
|
|
255
291
|
`${label} returned an empty completion (finish reason: ${finishReason || 'none given'}). ` +
|
|
256
|
-
'Check the model is fully loaded on the server.');
|
|
292
|
+
'Check the model is fully loaded on the server.', { usage });
|
|
257
293
|
}
|
|
258
294
|
|
|
259
295
|
/** Did the completion stop well short of the ceiling? Then the ceiling was not the problem. */
|
package/reasoning.js
ADDED
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* How much a model may think — the one rule, for every provider (0.27.0).
|
|
3
|
+
*
|
|
4
|
+
* THE APP DECIDES, PER CALL: `complete({ reasoning: 'off' | 'full' })`. Absent or null means 'off'.
|
|
5
|
+
* Before 0.27.0 only lmx honoured this; an LM Studio or OpenAI-compatible endpoint running Qwen3 or
|
|
6
|
+
* gpt-oss reasoned on EVERY call (its server default) and spent budgets sized for a quick answer,
|
|
7
|
+
* and the client stripped the option with a warning. Now each adapter translates the mode into what
|
|
8
|
+
* its model family understands:
|
|
9
|
+
*
|
|
10
|
+
* OpenAI-shaped (lmx, openai-compatible, lmstudio) — a request-body flag per model family.
|
|
11
|
+
* Anthropic — `thinking` / `output_config.effort` per model generation, because the API differs by
|
|
12
|
+
* generation: Claude Opus 5.5 and Fable cannot turn thinking off at all (effort is the only lever),
|
|
13
|
+
* Sonnet 5.5 turns it off with `between_tools`, 4.6 and older take an explicit on-switch, and
|
|
14
|
+
* several current models refuse `temperature` outright.
|
|
15
|
+
*
|
|
16
|
+
* 'off' IS "AS LITTLE AS THE FAMILY ALLOWS", not "zero": on a model that always thinks it is the
|
|
17
|
+
* lowest effort. An UNRECOGNISED family gets nothing — guessing a flag a model ignores wastes the
|
|
18
|
+
* budget silently, which is the failure this exists to avoid — and keeps the pre-0.27 request.
|
|
19
|
+
*
|
|
20
|
+
* A NAMED OPTION, NOT A RAW `extra`: the flag is merged after `extra`, so a stray passthrough can
|
|
21
|
+
* never turn thinking on and spend a budget sized for a quick answer.
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
'use strict';
|
|
25
|
+
|
|
26
|
+
const { AiError } = require('./error');
|
|
27
|
+
|
|
28
|
+
const REASONING_MODES = Object.freeze(['off', 'full']);
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Refuse a `reasoning` value this package does not define. `undefined`/`null` mean 'off'.
|
|
32
|
+
* Anything else — 'high', 'Full', true, '' — is a caller's mistake, and silently treating it as the
|
|
33
|
+
* default would hand a triage caller a thinking-off answer while it believed it had asked for more.
|
|
34
|
+
*/
|
|
35
|
+
function assertReasoning(value) {
|
|
36
|
+
if (value === undefined || value === null) return;
|
|
37
|
+
if (!REASONING_MODES.includes(value)) {
|
|
38
|
+
throw new AiError('refused',
|
|
39
|
+
`Unknown reasoning option ${JSON.stringify(value)}. The values are `
|
|
40
|
+
+ `${REASONING_MODES.map((m) => `'${m}'`).join(', ')}; leave it out for 'off'.`);
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
const isFull = (mode) => mode === 'full';
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* OpenAI-shaped families with a known flag, in match order. A TABLE rather than a chain of ifs so
|
|
48
|
+
* the README test can enumerate it: a family added here and not documented fails that test.
|
|
49
|
+
*/
|
|
50
|
+
const REASONING_FAMILIES = Object.freeze([
|
|
51
|
+
{ family: 'qwen', match: /qwen/, flag: (full) => ({ chat_template_kwargs: { enable_thinking: full } }) },
|
|
52
|
+
{ family: 'gpt-oss', match: /gpt-oss/, flag: (full) => ({ reasoning_effort: full ? 'high' : 'low' }) }
|
|
53
|
+
]);
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* LM STUDIO IGNORES chat_template_kwargs. Measured live 4 Oct 2026 (LM Studio, qwen/qwen3-1.7b):
|
|
57
|
+
* `enable_thinking: false` thought for 216 tokens, the same as `true`. What it DOES honour is
|
|
58
|
+
* `reasoning_effort`: 'none' turns Qwen thinking off (3 tokens, 0.9 s against 250 tokens, 14 s),
|
|
59
|
+
* and every other value (minimal/low/medium/high) means "on", ungraded. So on `provider: 'lmstudio'`
|
|
60
|
+
* Qwen gets reasoning_effort as well. NOT on plain openai-compatible: vLLM validates that field and
|
|
61
|
+
* may refuse 'none', and llama.cpp / vLLM already honour the template kwarg (lmx proves it daily).
|
|
62
|
+
* gpt-oss needs nothing extra — reasoning_effort is already its own flag.
|
|
63
|
+
*/
|
|
64
|
+
const LMSTUDIO_EXTRA = Object.freeze({
|
|
65
|
+
qwen: (full) => ({ reasoning_effort: full ? 'high' : 'none' })
|
|
66
|
+
});
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* ...AND THE DOCUMENTED SWITCH, TOO. LM Studio documents no per-request reasoning field on its
|
|
70
|
+
* OpenAI-compatible endpoint for Qwen (reasoning_effort is documented for gpt-oss only); what it does
|
|
71
|
+
* document, on the Qwen3 model pages, is Qwen's own soft switch: `/no_think` (or `/think`) in the
|
|
72
|
+
* prompt. Both are sent (decided with Petrus 4 Oct 2026): the body field is verified live but
|
|
73
|
+
* undocumented, the prompt switch is documented but only hybrid Qwen3 reads it — if a future LM
|
|
74
|
+
* Studio drops one, the other still holds. Appended to the system prompt by openai-compatible.
|
|
75
|
+
*/
|
|
76
|
+
const LMSTUDIO_PROMPT_SWITCH = Object.freeze({
|
|
77
|
+
qwen: (full) => (full ? '/think' : '/no_think')
|
|
78
|
+
});
|
|
79
|
+
|
|
80
|
+
/** The prompt switch for an LM Studio model and mode, or null. */
|
|
81
|
+
function lmstudioPromptSwitch(model, mode, provider) {
|
|
82
|
+
assertReasoning(mode);
|
|
83
|
+
if (provider !== 'lmstudio') return null;
|
|
84
|
+
const m = String(model || '').toLowerCase();
|
|
85
|
+
const f = REASONING_FAMILIES.find((x) => x.match.test(m));
|
|
86
|
+
const sw = f && LMSTUDIO_PROMPT_SWITCH[f.family];
|
|
87
|
+
return sw ? sw(isFull(mode)) : null;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* The OpenAI-shaped body flag for a model (path or name), mode and provider, or null for an unknown
|
|
92
|
+
* family. `provider` matters only for 'lmstudio' (see LMSTUDIO_EXTRA).
|
|
93
|
+
*/
|
|
94
|
+
function reasoningFor(model, mode, provider) {
|
|
95
|
+
assertReasoning(mode);
|
|
96
|
+
const m = String(model || '').toLowerCase();
|
|
97
|
+
const f = REASONING_FAMILIES.find((x) => x.match.test(m));
|
|
98
|
+
if (!f) return null;
|
|
99
|
+
const flag = f.flag(isFull(mode));
|
|
100
|
+
const extra = provider === 'lmstudio' && LMSTUDIO_EXTRA[f.family];
|
|
101
|
+
return extra ? { ...flag, ...extra(isFull(mode)) } : flag;
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* Anthropic generations, in match order (5-5 before 5). Each row says what 'off' and 'full' send and
|
|
106
|
+
* whether the model still accepts `temperature`.
|
|
107
|
+
*
|
|
108
|
+
* effort → output_config.effort
|
|
109
|
+
* thinking → the `thinking` object
|
|
110
|
+
* budget → `{ type: 'enabled', budget_tokens }` sized from maxTokens (pre-4.6 models)
|
|
111
|
+
* sampling → false: the model rejects temperature, so none is sent
|
|
112
|
+
*
|
|
113
|
+
* `examples` are real model ids that must land on their own row — the test checks each one, which is
|
|
114
|
+
* what catches a match-order mistake (opus-5 swallowing opus-5-5).
|
|
115
|
+
*
|
|
116
|
+
* Source: the Messages API thinking table (cached 2026-09-25). Opus 5 'off' is low effort rather
|
|
117
|
+
* than `disabled`, which Anthropic documents as leaking tool calls and thinking tags into text.
|
|
118
|
+
*/
|
|
119
|
+
const ANTHROPIC_FAMILIES = Object.freeze([
|
|
120
|
+
{ family: 'claude-fable / claude-opus-5-5', match: /fable|mythos|opus-5-5/,
|
|
121
|
+
examples: ['claude-opus-5-5', 'claude-fable-5-1', 'claude-fable-5', 'claude-mythos-5-1'],
|
|
122
|
+
off: { effort: 'low' }, full: { effort: 'high' }, sampling: false },
|
|
123
|
+
{ family: 'claude-opus-5', match: /opus-5(?![-.]?\d)/,
|
|
124
|
+
examples: ['claude-opus-5'],
|
|
125
|
+
off: { effort: 'low' }, full: { thinking: { type: 'adaptive' }, effort: 'high' }, sampling: false },
|
|
126
|
+
{ family: 'claude-sonnet-5-5', match: /sonnet-5-5/,
|
|
127
|
+
examples: ['claude-sonnet-5-5'],
|
|
128
|
+
off: { thinking: { type: 'between_tools' } }, full: { thinking: { type: 'adaptive' }, effort: 'high' }, sampling: false },
|
|
129
|
+
{ family: 'claude-sonnet-5', match: /sonnet-5(?![-.]?\d)/,
|
|
130
|
+
examples: ['claude-sonnet-5'],
|
|
131
|
+
off: { thinking: { type: 'disabled' } }, full: { thinking: { type: 'adaptive' }, effort: 'high' }, sampling: false },
|
|
132
|
+
{ family: 'claude-opus-4-7 / 4-8', match: /opus-4-[78]/,
|
|
133
|
+
examples: ['claude-opus-4-8', 'claude-opus-4-7'],
|
|
134
|
+
off: {}, full: { thinking: { type: 'adaptive' }, effort: 'high' }, sampling: false },
|
|
135
|
+
{ family: 'claude-*-4-6', match: /(opus|sonnet)-4-6/,
|
|
136
|
+
examples: ['claude-opus-4-6', 'claude-sonnet-4-6'],
|
|
137
|
+
off: {}, full: { thinking: { type: 'adaptive' }, effort: 'high' }, sampling: true },
|
|
138
|
+
{ family: 'claude-*-4-5 and older', match: /-4-5|-4-1|(opus|sonnet|haiku)-4(?![-.]?\d)|claude-3/,
|
|
139
|
+
examples: ['claude-haiku-4-5', 'claude-sonnet-4-5', 'claude-opus-4-5', 'claude-opus-4-1', 'claude-3-7-sonnet-latest'],
|
|
140
|
+
off: {}, full: { budget: true }, sampling: true }
|
|
141
|
+
]);
|
|
142
|
+
|
|
143
|
+
/** Thinking room added on top of the answer's own maxTokens when a pre-4.6 model is asked to think. */
|
|
144
|
+
const MIN_THINKING_BUDGET = 1024;
|
|
145
|
+
|
|
146
|
+
/**
|
|
147
|
+
* What to change on an Anthropic request body for this model and mode, or null for an unknown model
|
|
148
|
+
* (which keeps the pre-0.27 request: no thinking field, temperature as given).
|
|
149
|
+
*
|
|
150
|
+
* @returns {{family:string, thinking?:object, effort?:string, sendTemperature:boolean, maxTokens:number}|null}
|
|
151
|
+
*/
|
|
152
|
+
function anthropicReasoning(model, mode, maxTokens) {
|
|
153
|
+
assertReasoning(mode);
|
|
154
|
+
const m = String(model || '').toLowerCase();
|
|
155
|
+
const f = ANTHROPIC_FAMILIES.find((x) => x.match.test(m));
|
|
156
|
+
if (!f) return null;
|
|
157
|
+
const want = isFull(mode) ? f.full : f.off;
|
|
158
|
+
const out = { family: f.family, sendTemperature: f.sampling, maxTokens };
|
|
159
|
+
if (want.thinking) out.thinking = want.thinking;
|
|
160
|
+
if (want.effort) out.effort = want.effort;
|
|
161
|
+
if (want.budget) {
|
|
162
|
+
// The caller's maxTokens stays the ANSWER's room; the thinking budget is added on top, because
|
|
163
|
+
// budget_tokens must be below max_tokens and a budget carved out of the answer would starve it.
|
|
164
|
+
const budget = Math.max(MIN_THINKING_BUDGET, maxTokens);
|
|
165
|
+
out.thinking = { type: 'enabled', budget_tokens: budget };
|
|
166
|
+
out.maxTokens = maxTokens + budget;
|
|
167
|
+
// Extended thinking requires the default temperature on these models.
|
|
168
|
+
out.sendTemperature = false;
|
|
169
|
+
}
|
|
170
|
+
return out;
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
module.exports = {
|
|
174
|
+
REASONING_MODES, REASONING_FAMILIES, LMSTUDIO_EXTRA, LMSTUDIO_PROMPT_SWITCH, ANTHROPIC_FAMILIES, MIN_THINKING_BUDGET,
|
|
175
|
+
assertReasoning, reasoningFor, lmstudioPromptSwitch, anthropicReasoning
|
|
176
|
+
};
|