@aria-framework/ai 0.26.1 → 0.27.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@aria-framework/ai",
3
- "description": "Aria App Framework \u2014 AI module. A dependency-injected model seam (createAiClient) over several providers (LM Studio / OpenAI-compatible / Anthropic), with a fact-preservation guard, generic Polish and Generate writing engines, and a browser polish widget. Prompts and config stay in the consuming app.",
4
- "version": "0.26.1",
3
+ "description": "Aria App Framework — AI module. A dependency-injected model seam (createAiClient) over several providers (LM Studio / OpenAI-compatible / Anthropic), with a fact-preservation guard, generic Polish and Generate writing engines, and a browser polish widget. Prompts and config stay in the consuming app.",
4
+ "version": "0.27.1",
5
5
  "license": "UNLICENSED",
6
6
  "private": false,
7
7
  "publishConfig": {
@@ -9,6 +9,7 @@
9
9
  },
10
10
  "main": "index.js",
11
11
  "files": [
12
+ "CHANGELOG.md",
12
13
  "benchmark.js",
13
14
  "browser/ai-jobs.js",
14
15
  "browser/ai-panels.js",
@@ -29,6 +30,7 @@
29
30
  "providers/lmxDiscovery.js",
30
31
  "providers/lmxTransport.js",
31
32
  "providers/openai-compatible.js",
33
+ "reasoning.js",
32
34
  "speedStore.js",
33
35
  "untrusted.js",
34
36
  "usageStore.js",
@@ -47,7 +49,7 @@
47
49
  }
48
50
  },
49
51
  "scripts": {
50
- "test": "node test/smoke.js && node test/usageStore.js && node test/providerStore.js && node test/speedStore.js && node test/health.js && node test/listModels.js && node test/benchmark.js && node test/lmxDiscovery.js && node test/lmx.js && node test/lmxVerify.js && node test/lmxStore.js && node test/lmxStatus.js && node test/jobCard.js && node test/untrusted.js && node test/packaging.js && node test/views.js && node test/fenced.js && node test/polish.js && node test/reasoning.js && node test/usageOnFailure.js",
52
+ "test": "node test/smoke.js && node test/usageStore.js && node test/providerStore.js && node test/speedStore.js && node test/health.js && node test/listModels.js && node test/benchmark.js && node test/lmxDiscovery.js && node test/lmx.js && node test/lmxVerify.js && node test/lmxStore.js && node test/lmxStatus.js && node test/jobCard.js && node test/untrusted.js && node test/packaging.js && node test/views.js && node test/fenced.js && node test/polish.js && node test/reasoning.js && node test/reasoningProviders.js && node test/anthropicRequests.js && node test/usageOnFailure.js",
51
53
  "prepublishOnly": "node ../../test/packaging.js ai"
52
54
  },
53
55
  "devDependencies": {
@@ -19,6 +19,7 @@
19
19
  'use strict';
20
20
 
21
21
  const { AiError, fromFetchFailure, redact } = require('../error');
22
+ const { anthropicReasoning } = require('../reasoning');
22
23
 
23
24
  const API_VERSION = '2023-06-01';
24
25
  const DEFAULT_BASE = 'https://api.anthropic.com/v1';
@@ -60,6 +61,28 @@ async function complete(cfg, opts) {
60
61
  if (beta) headers['anthropic-beta'] = beta;
61
62
  }
62
63
 
64
+ // HOW MUCH CLAUDE MAY THINK (0.27.0) — per model generation, because the API differs by
65
+ // generation (see ../reasoning.js). Absent means 'off'. An unknown model keeps the pre-0.27 body.
66
+ const r = anthropicReasoning(cfg.model, opts.reasoning, body.max_tokens);
67
+ if (r) {
68
+ if (r.thinking) body.thinking = r.thinking;
69
+ // Merged into output_config, which may already carry the structured-output format.
70
+ if (r.effort) body.output_config = { ...(body.output_config || {}), effort: r.effort };
71
+ // Several current models reject temperature outright, and extended thinking on older ones
72
+ // requires the default. Sending it would be a 400, so it is left out.
73
+ if (!r.sendTemperature) delete body.temperature;
74
+ body.max_tokens = r.maxTokens;
75
+ // A MODEL WITH NO EXTENDED THINKING (0.27.1): nothing is sent, and 'full' is said to be
76
+ // unhonoured — a warning, as for an unknown model, never an error.
77
+ if (r.cannotThink && cfg.logger) {
78
+ cfg.logger.warn(`${label}: reasoning 'full' was asked for, but model "${cfg.model}" has no extended `
79
+ + 'thinking — sending no thinking setting; the model answers without it.');
80
+ }
81
+ } else if (opts.reasoning === 'full' && cfg.logger) {
82
+ cfg.logger.warn(`${label}: reasoning 'full' was asked for, but model "${cfg.model}" is not in the `
83
+ + 'known Claude generations — sending no thinking setting; the model runs on its own default.');
84
+ }
85
+
63
86
  const started = Date.now();
64
87
  const controller = new AbortController();
65
88
  const timer = setTimeout(() => controller.abort(), cfg.timeoutMs || 60000);
@@ -116,7 +139,7 @@ async function complete(cfg, opts) {
116
139
  const thought = (payload.content || []).some((b) => b && b.type === 'thinking');
117
140
  throw new AiError('bad_response', thought
118
141
  ? `${label} returned only its internal reasoning. Raise the reply limit and try again.`
119
- : `${label} returned no text (stop reason: ${payload.stop_reason || 'none given'}).`, { usage });
142
+ : `${label} returned no text (stop reason: ${payload.stop_reason || 'none given'}).`, { usage, model: payload.model });
120
143
  }
121
144
 
122
145
  let json = null;
@@ -128,10 +151,10 @@ async function complete(cfg, opts) {
128
151
  if (payload.stop_reason === 'max_tokens') {
129
152
  throw new AiError('bad_response',
130
153
  `${label} ran out of room mid-answer — the structured reply was cut off before it was ` +
131
- 'finished. A Regenerate usually succeeds.', { cause: err, usage });
154
+ 'finished. A Regenerate usually succeeds.', { cause: err, usage, model: payload.model });
132
155
  }
133
156
  throw new AiError('bad_response',
134
- `${label} was asked for structured output and returned text that will not parse.`, { cause: err, usage });
157
+ `${label} was asked for structured output and returned text that will not parse.`, { cause: err, usage, model: payload.model });
135
158
  }
136
159
  }
137
160
 
package/providers/lmx.js CHANGED
@@ -137,56 +137,15 @@ function skipError(name, res) {
137
137
  }
138
138
 
139
139
  /**
140
- * THE REASONING FLAG IS A PROPERTY OF THE MODEL, NOT OF THE ENGINE OR THE JOB.
141
- *
142
- * Both families reason before answering, and each needs a different instruction to stop. Getting it
143
- * wrong costs the whole budget and returns nothing — `content: ""` with `finish_reason: "length"`,
144
- * no error, and a retry produces the same nothing.
140
+ * THE REASONING FLAG IS A PROPERTY OF THE MODEL, NOT OF THE ENGINE.
145
141
  *
146
142
  * Read from the live document every call, never stored: point an engine at a different model and
147
- * the correct flag changes underneath a configuration that never changed.
148
- *
149
- * An UNRECOGNISED family gets NEITHER flag. Guessing would be worse than not guessing — sending a
150
- * Qwen argument to a model that ignores it wastes the budget silently, which is the failure this
151
- * exists to avoid.
152
- *
153
- * THE DEFAULT IS "THINK AS LITTLE AS THE FAMILY ALLOWS"; `reasoning: 'full'` (0.26.0) is the one
154
- * named way to ask for the opposite — Qwen thinking on, gpt-oss effort high — for judgement work
155
- * (triage) where a thinking-off answer measured as a different product. It is a NAMED per-call
156
- * option rather than a raw `extra`, because the forced flag deliberately outranks `extra`: a stray
157
- * passthrough must never be able to turn thinking on and spend a budget sized for a quick answer.
158
- */
159
- const REASONING_MODES = Object.freeze(['full']);
160
-
161
- /**
162
- * Refuse a `reasoning` value this package does not define. `undefined`/`null` mean the default.
163
- * Anything else — 'high', 'Full', true, '' — is a caller's mistake, and silently treating it as the
164
- * default would hand a triage caller a thinking-off answer while it believed it had asked for more.
165
- */
166
- function assertReasoning(value) {
167
- if (value === undefined || value === null) return;
168
- if (!REASONING_MODES.includes(value)) {
169
- throw new AiError('refused',
170
- `Unknown reasoning option ${JSON.stringify(value)}. The only value is `
171
- + `${REASONING_MODES.map((m) => `'${m}'`).join(', ')}; leave it out for the default.`);
172
- }
173
- }
174
-
175
- /**
176
- * The families with a known flag, in match order. A TABLE rather than a chain of ifs so the README
177
- * test can enumerate it: a family added here and not documented fails that test.
143
+ * the correct flag changes underneath a configuration that never changed. The table and the modes
144
+ * moved to ../reasoning.js in 0.27.0, when every provider started honouring `reasoning`; they are
145
+ * re-exported below so existing imports keep working. openai-compatible applies the flag, from the
146
+ * engine's model (`reasoningModel` on the engine config) — one place decides what is sent.
178
147
  */
179
- const REASONING_FAMILIES = Object.freeze([
180
- { family: 'qwen', match: /qwen/, flag: (full) => ({ chat_template_kwargs: { enable_thinking: full } }) },
181
- { family: 'gpt-oss', match: /gpt-oss/, flag: (full) => ({ reasoning_effort: full ? 'high' : 'low' }) }
182
- ]);
183
-
184
- function reasoningFor(modelPath, mode) {
185
- assertReasoning(mode);
186
- const m = String(modelPath || '').toLowerCase();
187
- const f = REASONING_FAMILIES.find((x) => x.match.test(m));
188
- return f ? f.flag(mode === 'full') : null;
189
- }
148
+ const { REASONING_MODES, REASONING_FAMILIES, assertReasoning, reasoningFor } = require('../reasoning');
190
149
 
191
150
  /** The config openai-compatible needs, once the address is known. */
192
151
  function engineConfig(cfg, url, engine) {
@@ -204,7 +163,10 @@ function engineConfig(cfg, url, engine) {
204
163
  : lmxTransport(cfg.lmx && cfg.lmx.ca),
205
164
  // SIZE AGAINST WHAT THE ENGINE REPORTS, not what somebody typed. maxInputTokens is PER SLOT,
206
165
  // because -c in llama.cpp is a pool divided across slots.
207
- contextTokens: (engine && engine.maxInputTokens) || cfg.contextTokens
166
+ contextTokens: (engine && engine.maxInputTokens) || cfg.contextTokens,
167
+ // THE MODEL THE FLAG IS CHOSEN FROM — the one the engine is running, not the endpoint's model
168
+ // name (an lmx endpoint stores none). openai-compatible reads this before cfg.model.
169
+ reasoningModel: engine && engine.model
208
170
  };
209
171
  }
210
172
 
@@ -254,14 +216,10 @@ async function complete(cfg, opts) {
254
216
  );
255
217
  }
256
218
 
257
- // The named option is consumed here; openai-compatible never sees it.
258
- const { reasoning: _consumed, ...rest } = opts || {};
259
- return openai.complete(engineConfig(cfg, res.url, res.engine), {
260
- ...rest,
261
- // Merged rather than replacing: a caller's own extras survive. The flag is spread LAST, so a
262
- // raw extra can never override it — only the named `reasoning` option changes what it says.
263
- extra: { ...(opts && opts.extra), ...(reasoning || {}) }
264
- }).catch((err) => { throw tagThrottle(err); });
219
+ // The mode goes on to openai-compatible, which merges the flag AFTER `extra` (0.27.0) — so a raw
220
+ // extra still cannot override it, and the option itself never reaches the request body.
221
+ return openai.complete(engineConfig(cfg, res.url, res.engine), opts || {})
222
+ .catch((err) => { throw tagThrottle(err); });
265
223
  }
266
224
 
267
225
  async function embed(cfg, texts) {
@@ -12,6 +12,7 @@
12
12
 
13
13
  'use strict';
14
14
 
15
+ const { reasoningFor, lmstudioPromptSwitch } = require('../reasoning');
15
16
  const { AiError, fromFetchFailure, redact } = require('../error');
16
17
 
17
18
  /**
@@ -65,6 +66,30 @@ async function complete(cfg, opts) {
65
66
  };
66
67
  }
67
68
 
69
+ // HOW MUCH THE MODEL MAY THINK (0.27.0), from the model family — the engine's model on lmx
70
+ // (`reasoningModel`), the configured name otherwise. Merged AFTER `extra`, so a raw passthrough
71
+ // cannot override it; only the named `reasoning` option changes it, and that option is never
72
+ // copied into the body itself. Absent means 'off': before 0.27.0 an LM Studio endpoint running
73
+ // Qwen3 reasoned on every call by default, spending budgets sized for a quick answer.
74
+ const flag = reasoningFor(cfg.reasoningModel || cfg.model, opts.reasoning, cfg.provider);
75
+ if (flag) Object.assign(body, flag);
76
+ else if (opts.reasoning === 'full' && cfg.logger && !cfg.lmx) {
77
+ // lmx warns for itself, naming the engine; this is the plain-endpoint case.
78
+ cfg.logger.warn(`${label}: reasoning 'full' was asked for, but no reasoning flag is known for model `
79
+ + `"${cfg.model}" — sending none; the model runs on its own default.`);
80
+ }
81
+
82
+ // LM STUDIO ALSO GETS QWEN'S DOCUMENTED PROMPT SWITCH (see reasoning.js LMSTUDIO_PROMPT_SWITCH),
83
+ // on the system prompt — appended to the caller's own, or as one of its own when there is none.
84
+ const promptSwitch = lmstudioPromptSwitch(cfg.reasoningModel || cfg.model, opts.reasoning, cfg.provider);
85
+ if (promptSwitch) {
86
+ if (messages.length && messages[0].role === 'system' && typeof messages[0].content === 'string') {
87
+ messages[0] = { role: 'system', content: `${messages[0].content}\n\n${promptSwitch}` };
88
+ } else {
89
+ messages.unshift({ role: 'system', content: promptSwitch });
90
+ }
91
+ }
92
+
68
93
  const started = Date.now();
69
94
  const controller = new AbortController();
70
95
  const timer = setTimeout(() => controller.abort(), cfg.timeoutMs || 60000);
@@ -139,6 +164,10 @@ async function complete(cfg, opts) {
139
164
  throw new AiError('bad_response', `${label} answered with an empty body.`);
140
165
  }
141
166
 
167
+ // THE MODEL THAT RAN (0.27.1): what the reply names, else — on lmx — the model the engine reports
168
+ // running (`reasoningModel`). Carried on errors for metering and on the result.
169
+ const ranModel = payload.model || cfg.reasoningModel;
170
+
142
171
  if (payload.error && !payload.choices) {
143
172
  const said = typeof payload.error === 'string'
144
173
  ? payload.error
@@ -173,7 +202,8 @@ async function complete(cfg, opts) {
173
202
  }
174
203
 
175
204
  if (!text) {
176
- throw emptyCompletion({ label, finishReason, reasoning, maxTokens: body.max_tokens, usage: payload.usage });
205
+ throw emptyCompletion({ label, finishReason, reasoning, maxTokens: body.max_tokens, usage: payload.usage,
206
+ model: ranModel });
177
207
  }
178
208
 
179
209
  // TOKENS WERE SPENT EVEN WHEN THE ANSWER IS UNUSABLE (0.26.1). Carried on every error raised after
@@ -198,17 +228,17 @@ async function complete(cfg, opts) {
198
228
  throw new AiError('bad_response',
199
229
  `${label} ran out of room mid-answer (${body.max_tokens} tokens) — the structured reply ` +
200
230
  'was cut off before it was finished. If this repeats, the schema may be letting the ' +
201
- 'model ramble; a Regenerate usually succeeds.', { cause: err, usage });
231
+ 'model ramble; a Regenerate usually succeeds.', { cause: err, usage, model: ranModel });
202
232
  }
203
233
  throw new AiError('bad_response',
204
- `${label} was asked for structured output and returned text that will not parse.`, { cause: err, usage });
234
+ `${label} was asked for structured output and returned text that will not parse.`, { cause: err, usage, model: ranModel });
205
235
  }
206
236
  }
207
237
 
208
238
  return {
209
239
  text,
210
240
  json,
211
- model: payload.model || cfg.model,
241
+ model: ranModel || cfg.model,
212
242
  usage,
213
243
  ms: Date.now() - started,
214
244
  finishReason,
@@ -246,7 +276,7 @@ function unclosedThinking(text) {
246
276
  * payload, so they are distinguished: the budget ran out, the budget went on thinking, or the server
247
277
  * genuinely said nothing.
248
278
  */
249
- function emptyCompletion({ label, finishReason, reasoning, maxTokens, usage: raw }) {
279
+ function emptyCompletion({ label, finishReason, reasoning, maxTokens, usage: raw, model }) {
250
280
  const spent = (raw && raw.completion_tokens) || 0;
251
281
  const usage = normaliseUsage(raw);
252
282
  if (finishReason === 'length' || (reasoning && !spentLeftRoom(spent, maxTokens))) {
@@ -255,16 +285,16 @@ function emptyCompletion({ label, finishReason, reasoning, maxTokens, usage: raw
255
285
  (reasoning ? ' on internal reasoning' : '') +
256
286
  '. This model thinks before it replies, so raise the reply limit (1024 is a sensible floor; ' +
257
287
  "4096 or more with reasoning: 'full').",
258
- { usage });
288
+ { usage, model });
259
289
  }
260
290
  if (reasoning) {
261
291
  return new AiError('bad_response',
262
292
  `${label} returned only its internal reasoning and no answer. Raise the reply limit and try again.`,
263
- { usage });
293
+ { usage, model });
264
294
  }
265
295
  return new AiError('bad_response',
266
296
  `${label} returned an empty completion (finish reason: ${finishReason || 'none given'}). ` +
267
- 'Check the model is fully loaded on the server.', { usage });
297
+ 'Check the model is fully loaded on the server.', { usage, model });
268
298
  }
269
299
 
270
300
  /** Did the completion stop well short of the ceiling? Then the ceiling was not the problem. */
package/reasoning.js ADDED
@@ -0,0 +1,223 @@
1
+ /**
2
+ * How much a model may think — the one rule, for every provider (0.27.0).
3
+ *
4
+ * THE APP DECIDES, PER CALL: `complete({ reasoning: 'off' | 'full' })`. Absent or null means 'off'.
5
+ * Before 0.27.0 only lmx honoured this; an LM Studio or OpenAI-compatible endpoint running Qwen3 or
6
+ * gpt-oss reasoned on EVERY call (its server default) and spent budgets sized for a quick answer,
7
+ * and the client stripped the option with a warning. Now each adapter translates the mode into what
8
+ * its model family understands:
9
+ *
10
+ * OpenAI-shaped (lmx, openai-compatible, lmstudio) — a request-body flag per model family.
11
+ * Anthropic — `thinking` / `output_config.effort` per model generation, because the API differs by
12
+ * generation: Claude Opus 5.5 and Fable cannot turn thinking off at all (effort is the only lever),
13
+ * Sonnet 5.5 turns it off with `between_tools`, 4.6 and older take an explicit on-switch, and
14
+ * several current models refuse `temperature` outright.
15
+ *
16
+ * 'off' IS "AS LITTLE AS THE FAMILY ALLOWS", not "zero": on a model that always thinks it is the
17
+ * lowest effort. An UNRECOGNISED family gets nothing — guessing a flag a model ignores wastes the
18
+ * budget silently, which is the failure this exists to avoid — and keeps the pre-0.27 request.
19
+ *
20
+ * A NAMED OPTION, NOT A RAW `extra`: the flag is merged after `extra`, so a stray passthrough can
21
+ * never turn thinking on and spend a budget sized for a quick answer.
22
+ */
23
+
24
+ 'use strict';
25
+
26
+ const { AiError } = require('./error');
27
+
28
+ const REASONING_MODES = Object.freeze(['off', 'full']);
29
+
30
+ /**
31
+ * Refuse a `reasoning` value this package does not define. `undefined`/`null` mean 'off'.
32
+ * Anything else — 'high', 'Full', true, '' — is a caller's mistake, and silently treating it as the
33
+ * default would hand a triage caller a thinking-off answer while it believed it had asked for more.
34
+ */
35
+ function assertReasoning(value) {
36
+ if (value === undefined || value === null) return;
37
+ if (!REASONING_MODES.includes(value)) {
38
+ throw new AiError('refused',
39
+ `Unknown reasoning option ${JSON.stringify(value)}. The values are `
40
+ + `${REASONING_MODES.map((m) => `'${m}'`).join(', ')}; leave it out for 'off'.`);
41
+ }
42
+ }
43
+
44
+ const isFull = (mode) => mode === 'full';
45
+
46
+ /**
47
+ * OpenAI-shaped families with a known flag, in match order. A TABLE rather than a chain of ifs so
48
+ * the README test can enumerate it: a family added here and not documented fails that test.
49
+ */
50
+ const REASONING_FAMILIES = Object.freeze([
51
+ { family: 'qwen', match: /qwen/, flag: (full) => ({ chat_template_kwargs: { enable_thinking: full } }) },
52
+ { family: 'gpt-oss', match: /gpt-oss/, flag: (full) => ({ reasoning_effort: full ? 'high' : 'low' }) }
53
+ ]);
54
+
55
+ /**
56
+ * LM STUDIO IGNORES chat_template_kwargs. Measured live 4 Oct 2026 (LM Studio, qwen/qwen3-1.7b):
57
+ * `enable_thinking: false` thought for 216 tokens, the same as `true`. What it DOES honour is
58
+ * `reasoning_effort`: 'none' turns Qwen thinking off (3 tokens, 0.9 s against 250 tokens, 14 s),
59
+ * and every other value (minimal/low/medium/high) means "on", ungraded. So on `provider: 'lmstudio'`
60
+ * Qwen gets reasoning_effort as well. NOT on plain openai-compatible: vLLM validates that field and
61
+ * may refuse 'none', and llama.cpp / vLLM already honour the template kwarg (lmx proves it daily).
62
+ * gpt-oss needs nothing extra — reasoning_effort is already its own flag.
63
+ */
64
+ const LMSTUDIO_EXTRA = Object.freeze({
65
+ qwen: (full) => ({ reasoning_effort: full ? 'high' : 'none' })
66
+ });
67
+
68
+ /**
69
+ * ...AND THE DOCUMENTED SWITCH, TOO. LM Studio documents no per-request reasoning field on its
70
+ * OpenAI-compatible endpoint for Qwen (reasoning_effort is documented for gpt-oss only); what it does
71
+ * document, on the Qwen3 model pages, is Qwen's own soft switch: `/no_think` (or `/think`) in the
72
+ * prompt. Both are sent (decided with Petrus 4 Oct 2026): the body field is verified live but
73
+ * undocumented, the prompt switch is documented but only hybrid Qwen3 reads it — if a future LM
74
+ * Studio drops one, the other still holds. Appended to the system prompt by openai-compatible.
75
+ */
76
+ const LMSTUDIO_PROMPT_SWITCH = Object.freeze({
77
+ qwen: (full) => (full ? '/think' : '/no_think')
78
+ });
79
+
80
+ /** The prompt switch for an LM Studio model and mode, or null. */
81
+ function lmstudioPromptSwitch(model, mode, provider) {
82
+ assertReasoning(mode);
83
+ if (provider !== 'lmstudio') return null;
84
+ const m = String(model || '').toLowerCase();
85
+ const f = REASONING_FAMILIES.find((x) => x.match.test(m));
86
+ const sw = f && LMSTUDIO_PROMPT_SWITCH[f.family];
87
+ return sw ? sw(isFull(mode)) : null;
88
+ }
89
+
90
+ /**
91
+ * The OpenAI-shaped body flag for a model (path or name), mode and provider, or null for an unknown
92
+ * family. `provider` matters only for 'lmstudio' (see LMSTUDIO_EXTRA).
93
+ */
94
+ function reasoningFor(model, mode, provider) {
95
+ assertReasoning(mode);
96
+ const m = String(model || '').toLowerCase();
97
+ const f = REASONING_FAMILIES.find((x) => x.match.test(m));
98
+ if (!f) return null;
99
+ const flag = f.flag(isFull(mode));
100
+ const extra = provider === 'lmstudio' && LMSTUDIO_EXTRA[f.family];
101
+ return extra ? { ...flag, ...extra(isFull(mode)) } : flag;
102
+ }
103
+
104
+ /**
105
+ * Anthropic generations, in match order (5-5 before 5). Each row says what 'off' and 'full' send and
106
+ * whether the model still accepts `temperature`.
107
+ *
108
+ * effort → output_config.effort
109
+ * thinking → the `thinking` object
110
+ * budget → `{ type: 'enabled', budget_tokens }` sized from maxTokens (pre-4.6 models)
111
+ * maxOutput → the model's output limit; a budget row never sends max_tokens above it (0.27.1)
112
+ * cannotThink → the model has no extended thinking: 'full' sends nothing and is warned about,
113
+ * exactly as for an unknown model (0.27.1)
114
+ * sampling → false: the model rejects temperature, so none is sent
115
+ *
116
+ * `examples` are real model ids that must land on their own row — the test checks each one, which is
117
+ * what catches a match-order mistake (opus-5 swallowing opus-5-5).
118
+ *
119
+ * Source: the Messages API thinking table (cached 2026-09-25). Opus 5 'off' is low effort rather
120
+ * than `disabled`, which Anthropic documents as leaking tool calls and thinking tags into text.
121
+ */
122
+ const ANTHROPIC_FAMILIES = Object.freeze([
123
+ { family: 'claude-fable / claude-opus-5-5', match: /fable|mythos|opus-5-5/,
124
+ examples: ['claude-opus-5-5', 'claude-fable-5-1', 'claude-fable-5', 'claude-mythos-5-1'],
125
+ off: { effort: 'low' }, full: { effort: 'high' }, sampling: false },
126
+ { family: 'claude-opus-5', match: /opus-5(?![-.]?\d)/,
127
+ examples: ['claude-opus-5'],
128
+ off: { effort: 'low' }, full: { thinking: { type: 'adaptive' }, effort: 'high' }, sampling: false },
129
+ { family: 'claude-sonnet-5-5', match: /sonnet-5-5/,
130
+ examples: ['claude-sonnet-5-5'],
131
+ off: { thinking: { type: 'between_tools' } }, full: { thinking: { type: 'adaptive' }, effort: 'high' }, sampling: false },
132
+ { family: 'claude-sonnet-5', match: /sonnet-5(?![-.]?\d)/,
133
+ examples: ['claude-sonnet-5'],
134
+ off: { thinking: { type: 'disabled' } }, full: { thinking: { type: 'adaptive' }, effort: 'high' }, sampling: false },
135
+ { family: 'claude-opus-4-7 / 4-8', match: /opus-4-[78]/,
136
+ examples: ['claude-opus-4-8', 'claude-opus-4-7'],
137
+ off: {}, full: { thinking: { type: 'adaptive' }, effort: 'high' }, sampling: false },
138
+ { family: 'claude-*-4-6', match: /(opus|sonnet)-4-6/,
139
+ examples: ['claude-opus-4-6', 'claude-sonnet-4-6'],
140
+ off: {}, full: { thinking: { type: 'adaptive' }, effort: 'high' }, sampling: true },
141
+ // PRE-4.6, ONE ROW PER OUTPUT LIMIT (0.27.1). 0.27.0 had a single row and doubled max_tokens on
142
+ // 'full' with no ceiling — 80000 asked of Haiku 4.5, which allows 64000, is a 400. Limits from the
143
+ // model tables (cached 2026-09-25). Dated ids (claude-opus-4-20250514) and the -0 aliases land here
144
+ // too; 0.27.0 matched neither, so 'full' on them sent no thinking and warned "unknown model".
145
+ { family: 'claude-*-4-5', match: /-4-5/,
146
+ examples: ['claude-haiku-4-5', 'claude-sonnet-4-5', 'claude-opus-4-5', 'claude-sonnet-4-5-20250929'],
147
+ off: {}, full: { budget: true }, maxOutput: 64000, sampling: true },
148
+ { family: 'claude-opus-4-1 / claude-opus-4', match: /opus-4-1|opus-4(?:-0|-\d{8})?(?![-.]?\d)/,
149
+ examples: ['claude-opus-4-1', 'claude-opus-4-1-20250805', 'claude-opus-4-0', 'claude-opus-4-20250514'],
150
+ off: {}, full: { budget: true }, maxOutput: 32000, sampling: true },
151
+ { family: 'claude-sonnet-4', match: /sonnet-4(?:-0|-\d{8})?(?![-.]?\d)/,
152
+ examples: ['claude-sonnet-4-0', 'claude-sonnet-4-20250514'],
153
+ off: {}, full: { budget: true }, maxOutput: 64000, sampling: true },
154
+ { family: 'claude-3-7-sonnet', match: /claude-3-7/,
155
+ examples: ['claude-3-7-sonnet-latest', 'claude-3-7-sonnet-20250219'],
156
+ off: {}, full: { budget: true }, maxOutput: 64000, sampling: true },
157
+ // EVERY OTHER CLAUDE 3 HAS NO EXTENDED THINKING (0.27.1). 0.27.0's /claude-3/ sent them
158
+ // `thinking: enabled` on 'full', which the API refuses. Now they keep the pre-0.27 request and
159
+ // 'full' is a warning — the same as a model this table does not know — never an error: the caller
160
+ // asked for more thought, not for the call to fail.
161
+ { family: 'claude-3 without extended thinking', match: /claude-3/,
162
+ examples: ['claude-3-5-haiku-20241022', 'claude-3-5-haiku-latest', 'claude-3-haiku-20240307',
163
+ 'claude-3-opus-20240229', 'claude-3-5-sonnet-20241022'],
164
+ off: {}, full: {}, cannotThink: true, sampling: true }
165
+ ]);
166
+
167
+ /** Thinking room added on top of the answer's own maxTokens when a pre-4.6 model is asked to think. */
168
+ const MIN_THINKING_BUDGET = 1024;
169
+
170
+ /**
171
+ * maxTokens as a positive integer, or refused (0.27.1). A fraction is floored; anything that is not
172
+ * a finite number of at least 1 is the caller's bug. A string "2048" used to turn the budget sum into
173
+ * a string concatenation that Math.min read as the model's whole limit, and a negative value left
174
+ * budget_tokens above max_tokens — both sent without complaint.
175
+ */
176
+ function positiveTokens(value) {
177
+ const n = typeof value === 'number' && Number.isFinite(value) ? Math.floor(value) : NaN;
178
+ if (!(n >= 1)) {
179
+ const shown = typeof value === 'string' ? JSON.stringify(value) : String(value);
180
+ throw new AiError('refused', `maxTokens must be a positive number of tokens; got ${shown}.`);
181
+ }
182
+ return n;
183
+ }
184
+
185
+ /**
186
+ * What to change on an Anthropic request body for this model and mode, or null for an unknown model
187
+ * (which keeps the pre-0.27 request: no thinking field, temperature as given). `cannotThink` is set
188
+ * when 'full' was asked of a model with no extended thinking, so the adapter can say so.
189
+ *
190
+ * @returns {{family:string, thinking?:object, effort?:string, sendTemperature:boolean, maxTokens:number,
191
+ * cannotThink?:boolean}|null}
192
+ */
193
+ function anthropicReasoning(model, mode, maxTokensIn) {
194
+ assertReasoning(mode);
195
+ const maxTokens = positiveTokens(maxTokensIn);
196
+ const m = String(model || '').toLowerCase();
197
+ const f = ANTHROPIC_FAMILIES.find((x) => x.match.test(m));
198
+ if (!f) return null;
199
+ const want = isFull(mode) ? f.full : f.off;
200
+ const out = { family: f.family, sendTemperature: f.sampling, maxTokens };
201
+ if (want.thinking) out.thinking = want.thinking;
202
+ if (want.effort) out.effort = want.effort;
203
+ if (isFull(mode) && f.cannotThink) out.cannotThink = true;
204
+ if (want.budget) {
205
+ // The caller's maxTokens stays the ANSWER's room; the thinking budget is added on top, because
206
+ // budget_tokens must be below max_tokens and a budget carved out of the answer would starve it.
207
+ // CAPPED AT THE MODEL'S OUTPUT LIMIT (0.27.1): max_tokens counts the thinking too, and a request
208
+ // past the limit is refused. Near the limit the thinking room shrinks first, down to the API's
209
+ // minimum of 1024; only then does the answer's room give way.
210
+ const total = Math.min(maxTokens + Math.max(MIN_THINKING_BUDGET, maxTokens), f.maxOutput);
211
+ const budget = Math.max(MIN_THINKING_BUDGET, total - maxTokens);
212
+ out.thinking = { type: 'enabled', budget_tokens: budget };
213
+ out.maxTokens = total;
214
+ // Extended thinking requires the default temperature on these models.
215
+ out.sendTemperature = false;
216
+ }
217
+ return out;
218
+ }
219
+
220
+ module.exports = {
221
+ REASONING_MODES, REASONING_FAMILIES, LMSTUDIO_EXTRA, LMSTUDIO_PROMPT_SWITCH, ANTHROPIC_FAMILIES, MIN_THINKING_BUDGET,
222
+ assertReasoning, reasoningFor, lmstudioPromptSwitch, anthropicReasoning
223
+ };