@aria-framework/ai 0.26.1 → 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@aria-framework/ai",
3
- "description": "Aria App Framework \u2014 AI module. A dependency-injected model seam (createAiClient) over several providers (LM Studio / OpenAI-compatible / Anthropic), with a fact-preservation guard, generic Polish and Generate writing engines, and a browser polish widget. Prompts and config stay in the consuming app.",
4
- "version": "0.26.1",
3
+ "description": "Aria App Framework — AI module. A dependency-injected model seam (createAiClient) over several providers (LM Studio / OpenAI-compatible / Anthropic), with a fact-preservation guard, generic Polish and Generate writing engines, and a browser polish widget. Prompts and config stay in the consuming app.",
4
+ "version": "0.27.0",
5
5
  "license": "UNLICENSED",
6
6
  "private": false,
7
7
  "publishConfig": {
@@ -29,6 +29,7 @@
29
29
  "providers/lmxDiscovery.js",
30
30
  "providers/lmxTransport.js",
31
31
  "providers/openai-compatible.js",
32
+ "reasoning.js",
32
33
  "speedStore.js",
33
34
  "untrusted.js",
34
35
  "usageStore.js",
@@ -47,7 +48,7 @@
47
48
  }
48
49
  },
49
50
  "scripts": {
50
- "test": "node test/smoke.js && node test/usageStore.js && node test/providerStore.js && node test/speedStore.js && node test/health.js && node test/listModels.js && node test/benchmark.js && node test/lmxDiscovery.js && node test/lmx.js && node test/lmxVerify.js && node test/lmxStore.js && node test/lmxStatus.js && node test/jobCard.js && node test/untrusted.js && node test/packaging.js && node test/views.js && node test/fenced.js && node test/polish.js && node test/reasoning.js && node test/usageOnFailure.js",
51
+ "test": "node test/smoke.js && node test/usageStore.js && node test/providerStore.js && node test/speedStore.js && node test/health.js && node test/listModels.js && node test/benchmark.js && node test/lmxDiscovery.js && node test/lmx.js && node test/lmxVerify.js && node test/lmxStore.js && node test/lmxStatus.js && node test/jobCard.js && node test/untrusted.js && node test/packaging.js && node test/views.js && node test/fenced.js && node test/polish.js && node test/reasoning.js && node test/reasoningProviders.js && node test/usageOnFailure.js",
51
52
  "prepublishOnly": "node ../../test/packaging.js ai"
52
53
  },
53
54
  "devDependencies": {
@@ -19,6 +19,7 @@
19
19
  'use strict';
20
20
 
21
21
  const { AiError, fromFetchFailure, redact } = require('../error');
22
+ const { anthropicReasoning } = require('../reasoning');
22
23
 
23
24
  const API_VERSION = '2023-06-01';
24
25
  const DEFAULT_BASE = 'https://api.anthropic.com/v1';
@@ -60,6 +61,22 @@ async function complete(cfg, opts) {
60
61
  if (beta) headers['anthropic-beta'] = beta;
61
62
  }
62
63
 
64
+ // HOW MUCH CLAUDE MAY THINK (0.27.0) — per model generation, because the API differs by
65
+ // generation (see ../reasoning.js). Absent means 'off'. An unknown model keeps the pre-0.27 body.
66
+ const r = anthropicReasoning(cfg.model, opts.reasoning, body.max_tokens);
67
+ if (r) {
68
+ if (r.thinking) body.thinking = r.thinking;
69
+ // Merged into output_config, which may already carry the structured-output format.
70
+ if (r.effort) body.output_config = { ...(body.output_config || {}), effort: r.effort };
71
+ // Several current models reject temperature outright, and extended thinking on older ones
72
+ // requires the default. Sending it would be a 400, so it is left out.
73
+ if (!r.sendTemperature) delete body.temperature;
74
+ body.max_tokens = r.maxTokens;
75
+ } else if (opts.reasoning === 'full' && cfg.logger) {
76
+ cfg.logger.warn(`${label}: reasoning 'full' was asked for, but model "${cfg.model}" is not in the `
77
+ + 'known Claude generations — sending no thinking setting; the model runs on its own default.');
78
+ }
79
+
63
80
  const started = Date.now();
64
81
  const controller = new AbortController();
65
82
  const timer = setTimeout(() => controller.abort(), cfg.timeoutMs || 60000);
package/providers/lmx.js CHANGED
@@ -137,56 +137,15 @@ function skipError(name, res) {
137
137
  }
138
138
 
139
139
  /**
140
- * THE REASONING FLAG IS A PROPERTY OF THE MODEL, NOT OF THE ENGINE OR THE JOB.
141
- *
142
- * Both families reason before answering, and each needs a different instruction to stop. Getting it
143
- * wrong costs the whole budget and returns nothing — `content: ""` with `finish_reason: "length"`,
144
- * no error, and a retry produces the same nothing.
140
+ * THE REASONING FLAG IS A PROPERTY OF THE MODEL, NOT OF THE ENGINE.
145
141
  *
146
142
  * Read from the live document every call, never stored: point an engine at a different model and
147
- * the correct flag changes underneath a configuration that never changed.
148
- *
149
- * An UNRECOGNISED family gets NEITHER flag. Guessing would be worse than not guessing — sending a
150
- * Qwen argument to a model that ignores it wastes the budget silently, which is the failure this
151
- * exists to avoid.
152
- *
153
- * THE DEFAULT IS "THINK AS LITTLE AS THE FAMILY ALLOWS"; `reasoning: 'full'` (0.26.0) is the one
154
- * named way to ask for the opposite — Qwen thinking on, gpt-oss effort high — for judgement work
155
- * (triage) where a thinking-off answer measured as a different product. It is a NAMED per-call
156
- * option rather than a raw `extra`, because the forced flag deliberately outranks `extra`: a stray
157
- * passthrough must never be able to turn thinking on and spend a budget sized for a quick answer.
158
- */
159
- const REASONING_MODES = Object.freeze(['full']);
160
-
161
- /**
162
- * Refuse a `reasoning` value this package does not define. `undefined`/`null` mean the default.
163
- * Anything else — 'high', 'Full', true, '' — is a caller's mistake, and silently treating it as the
164
- * default would hand a triage caller a thinking-off answer while it believed it had asked for more.
165
- */
166
- function assertReasoning(value) {
167
- if (value === undefined || value === null) return;
168
- if (!REASONING_MODES.includes(value)) {
169
- throw new AiError('refused',
170
- `Unknown reasoning option ${JSON.stringify(value)}. The only value is `
171
- + `${REASONING_MODES.map((m) => `'${m}'`).join(', ')}; leave it out for the default.`);
172
- }
173
- }
174
-
175
- /**
176
- * The families with a known flag, in match order. A TABLE rather than a chain of ifs so the README
177
- * test can enumerate it: a family added here and not documented fails that test.
143
+ * the correct flag changes underneath a configuration that never changed. The table and the modes
144
+ * moved to ../reasoning.js in 0.27.0, when every provider started honouring `reasoning`; they are
145
+ * re-exported below so existing imports keep working. openai-compatible applies the flag, from the
146
+ * engine's model (`reasoningModel` on the engine config) — one place decides what is sent.
178
147
  */
179
- const REASONING_FAMILIES = Object.freeze([
180
- { family: 'qwen', match: /qwen/, flag: (full) => ({ chat_template_kwargs: { enable_thinking: full } }) },
181
- { family: 'gpt-oss', match: /gpt-oss/, flag: (full) => ({ reasoning_effort: full ? 'high' : 'low' }) }
182
- ]);
183
-
184
- function reasoningFor(modelPath, mode) {
185
- assertReasoning(mode);
186
- const m = String(modelPath || '').toLowerCase();
187
- const f = REASONING_FAMILIES.find((x) => x.match.test(m));
188
- return f ? f.flag(mode === 'full') : null;
189
- }
148
+ const { REASONING_MODES, REASONING_FAMILIES, assertReasoning, reasoningFor } = require('../reasoning');
190
149
 
191
150
  /** The config openai-compatible needs, once the address is known. */
192
151
  function engineConfig(cfg, url, engine) {
@@ -204,7 +163,10 @@ function engineConfig(cfg, url, engine) {
204
163
  : lmxTransport(cfg.lmx && cfg.lmx.ca),
205
164
  // SIZE AGAINST WHAT THE ENGINE REPORTS, not what somebody typed. maxInputTokens is PER SLOT,
206
165
  // because -c in llama.cpp is a pool divided across slots.
207
- contextTokens: (engine && engine.maxInputTokens) || cfg.contextTokens
166
+ contextTokens: (engine && engine.maxInputTokens) || cfg.contextTokens,
167
+ // THE MODEL THE FLAG IS CHOSEN FROM — the one the engine is running, not the endpoint's model
168
+ // name (an lmx endpoint stores none). openai-compatible reads this before cfg.model.
169
+ reasoningModel: engine && engine.model
208
170
  };
209
171
  }
210
172
 
@@ -254,14 +216,10 @@ async function complete(cfg, opts) {
254
216
  );
255
217
  }
256
218
 
257
- // The named option is consumed here; openai-compatible never sees it.
258
- const { reasoning: _consumed, ...rest } = opts || {};
259
- return openai.complete(engineConfig(cfg, res.url, res.engine), {
260
- ...rest,
261
- // Merged rather than replacing: a caller's own extras survive. The flag is spread LAST, so a
262
- // raw extra can never override it — only the named `reasoning` option changes what it says.
263
- extra: { ...(opts && opts.extra), ...(reasoning || {}) }
264
- }).catch((err) => { throw tagThrottle(err); });
219
+ // The mode goes on to openai-compatible, which merges the flag AFTER `extra` (0.27.0) — so a raw
220
+ // extra still cannot override it, and the option itself never reaches the request body.
221
+ return openai.complete(engineConfig(cfg, res.url, res.engine), opts || {})
222
+ .catch((err) => { throw tagThrottle(err); });
265
223
  }
266
224
 
267
225
  async function embed(cfg, texts) {
@@ -12,6 +12,7 @@
12
12
 
13
13
  'use strict';
14
14
 
15
+ const { reasoningFor, lmstudioPromptSwitch } = require('../reasoning');
15
16
  const { AiError, fromFetchFailure, redact } = require('../error');
16
17
 
17
18
  /**
@@ -65,6 +66,30 @@ async function complete(cfg, opts) {
65
66
  };
66
67
  }
67
68
 
69
+ // HOW MUCH THE MODEL MAY THINK (0.27.0), from the model family — the engine's model on lmx
70
+ // (`reasoningModel`), the configured name otherwise. Merged AFTER `extra`, so a raw passthrough
71
+ // cannot override it; only the named `reasoning` option changes it, and that option is never
72
+ // copied into the body itself. Absent means 'off': before 0.27.0 an LM Studio endpoint running
73
+ // Qwen3 reasoned on every call by default, spending budgets sized for a quick answer.
74
+ const flag = reasoningFor(cfg.reasoningModel || cfg.model, opts.reasoning, cfg.provider);
75
+ if (flag) Object.assign(body, flag);
76
+ else if (opts.reasoning === 'full' && cfg.logger && !cfg.lmx) {
77
+ // lmx warns for itself, naming the engine; this is the plain-endpoint case.
78
+ cfg.logger.warn(`${label}: reasoning 'full' was asked for, but no reasoning flag is known for model `
79
+ + `"${cfg.model}" — sending none; the model runs on its own default.`);
80
+ }
81
+
82
+ // LM STUDIO ALSO GETS QWEN'S DOCUMENTED PROMPT SWITCH (see reasoning.js LMSTUDIO_PROMPT_SWITCH),
83
+ // on the system prompt — appended to the caller's own, or as one of its own when there is none.
84
+ const promptSwitch = lmstudioPromptSwitch(cfg.reasoningModel || cfg.model, opts.reasoning, cfg.provider);
85
+ if (promptSwitch) {
86
+ if (messages.length && messages[0].role === 'system' && typeof messages[0].content === 'string') {
87
+ messages[0] = { role: 'system', content: `${messages[0].content}\n\n${promptSwitch}` };
88
+ } else {
89
+ messages.unshift({ role: 'system', content: promptSwitch });
90
+ }
91
+ }
92
+
68
93
  const started = Date.now();
69
94
  const controller = new AbortController();
70
95
  const timer = setTimeout(() => controller.abort(), cfg.timeoutMs || 60000);
package/reasoning.js ADDED
@@ -0,0 +1,176 @@
1
+ /**
2
+ * How much a model may think — the one rule, for every provider (0.27.0).
3
+ *
4
+ * THE APP DECIDES, PER CALL: `complete({ reasoning: 'off' | 'full' })`. Absent or null means 'off'.
5
+ * Before 0.27.0 only lmx honoured this; an LM Studio or OpenAI-compatible endpoint running Qwen3 or
6
+ * gpt-oss reasoned on EVERY call (its server default) and spent budgets sized for a quick answer,
7
+ * and the client stripped the option with a warning. Now each adapter translates the mode into what
8
+ * its model family understands:
9
+ *
10
+ * OpenAI-shaped (lmx, openai-compatible, lmstudio) — a request-body flag per model family.
11
+ * Anthropic — `thinking` / `output_config.effort` per model generation, because the API differs by
12
+ * generation: Claude Opus 5.5 and Fable cannot turn thinking off at all (effort is the only lever),
13
+ * Sonnet 5.5 turns it off with `between_tools`, 4.6 and older take an explicit on-switch, and
14
+ * several current models refuse `temperature` outright.
15
+ *
16
+ * 'off' IS "AS LITTLE AS THE FAMILY ALLOWS", not "zero": on a model that always thinks it is the
17
+ * lowest effort. An UNRECOGNISED family gets nothing — guessing a flag a model ignores wastes the
18
+ * budget silently, which is the failure this exists to avoid — and keeps the pre-0.27 request.
19
+ *
20
+ * A NAMED OPTION, NOT A RAW `extra`: the flag is merged after `extra`, so a stray passthrough can
21
+ * never turn thinking on and spend a budget sized for a quick answer.
22
+ */
23
+
24
+ 'use strict';
25
+
26
+ const { AiError } = require('./error');
27
+
28
+ const REASONING_MODES = Object.freeze(['off', 'full']);
29
+
30
+ /**
31
+ * Refuse a `reasoning` value this package does not define. `undefined`/`null` mean 'off'.
32
+ * Anything else — 'high', 'Full', true, '' — is a caller's mistake, and silently treating it as the
33
+ * default would hand a triage caller a thinking-off answer while it believed it had asked for more.
34
+ */
35
+ function assertReasoning(value) {
36
+ if (value === undefined || value === null) return;
37
+ if (!REASONING_MODES.includes(value)) {
38
+ throw new AiError('refused',
39
+ `Unknown reasoning option ${JSON.stringify(value)}. The values are `
40
+ + `${REASONING_MODES.map((m) => `'${m}'`).join(', ')}; leave it out for 'off'.`);
41
+ }
42
+ }
43
+
44
+ const isFull = (mode) => mode === 'full';
45
+
46
+ /**
47
+ * OpenAI-shaped families with a known flag, in match order. A TABLE rather than a chain of ifs so
48
+ * the README test can enumerate it: a family added here and not documented fails that test.
49
+ */
50
+ const REASONING_FAMILIES = Object.freeze([
51
+ { family: 'qwen', match: /qwen/, flag: (full) => ({ chat_template_kwargs: { enable_thinking: full } }) },
52
+ { family: 'gpt-oss', match: /gpt-oss/, flag: (full) => ({ reasoning_effort: full ? 'high' : 'low' }) }
53
+ ]);
54
+
55
+ /**
56
+ * LM STUDIO IGNORES chat_template_kwargs. Measured live 4 Oct 2026 (LM Studio, qwen/qwen3-1.7b):
57
+ * `enable_thinking: false` thought for 216 tokens, the same as `true`. What it DOES honour is
58
+ * `reasoning_effort`: 'none' turns Qwen thinking off (3 tokens, 0.9 s against 250 tokens, 14 s),
59
+ * and every other value (minimal/low/medium/high) means "on", ungraded. So on `provider: 'lmstudio'`
60
+ * Qwen gets reasoning_effort as well. NOT on plain openai-compatible: vLLM validates that field and
61
+ * may refuse 'none', and llama.cpp / vLLM already honour the template kwarg (lmx proves it daily).
62
+ * gpt-oss needs nothing extra — reasoning_effort is already its own flag.
63
+ */
64
+ const LMSTUDIO_EXTRA = Object.freeze({
65
+ qwen: (full) => ({ reasoning_effort: full ? 'high' : 'none' })
66
+ });
67
+
68
+ /**
69
+ * ...AND THE DOCUMENTED SWITCH, TOO. LM Studio documents no per-request reasoning field on its
70
+ * OpenAI-compatible endpoint for Qwen (reasoning_effort is documented for gpt-oss only); what it does
71
+ * document, on the Qwen3 model pages, is Qwen's own soft switch: `/no_think` (or `/think`) in the
72
+ * prompt. Both are sent (decided with Petrus 4 Oct 2026): the body field is verified live but
73
+ * undocumented, the prompt switch is documented but only hybrid Qwen3 reads it — if a future LM
74
+ * Studio drops one, the other still holds. Appended to the system prompt by openai-compatible.
75
+ */
76
+ const LMSTUDIO_PROMPT_SWITCH = Object.freeze({
77
+ qwen: (full) => (full ? '/think' : '/no_think')
78
+ });
79
+
80
+ /** The prompt switch for an LM Studio model and mode, or null. */
81
+ function lmstudioPromptSwitch(model, mode, provider) {
82
+ assertReasoning(mode);
83
+ if (provider !== 'lmstudio') return null;
84
+ const m = String(model || '').toLowerCase();
85
+ const f = REASONING_FAMILIES.find((x) => x.match.test(m));
86
+ const sw = f && LMSTUDIO_PROMPT_SWITCH[f.family];
87
+ return sw ? sw(isFull(mode)) : null;
88
+ }
89
+
90
+ /**
91
+ * The OpenAI-shaped body flag for a model (path or name), mode and provider, or null for an unknown
92
+ * family. `provider` matters only for 'lmstudio' (see LMSTUDIO_EXTRA).
93
+ */
94
+ function reasoningFor(model, mode, provider) {
95
+ assertReasoning(mode);
96
+ const m = String(model || '').toLowerCase();
97
+ const f = REASONING_FAMILIES.find((x) => x.match.test(m));
98
+ if (!f) return null;
99
+ const flag = f.flag(isFull(mode));
100
+ const extra = provider === 'lmstudio' && LMSTUDIO_EXTRA[f.family];
101
+ return extra ? { ...flag, ...extra(isFull(mode)) } : flag;
102
+ }
103
+
104
+ /**
105
+ * Anthropic generations, in match order (5-5 before 5). Each row says what 'off' and 'full' send and
106
+ * whether the model still accepts `temperature`.
107
+ *
108
+ * effort → output_config.effort
109
+ * thinking → the `thinking` object
110
+ * budget → `{ type: 'enabled', budget_tokens }` sized from maxTokens (pre-4.6 models)
111
+ * sampling → false: the model rejects temperature, so none is sent
112
+ *
113
+ * `examples` are real model ids that must land on their own row — the test checks each one, which is
114
+ * what catches a match-order mistake (opus-5 swallowing opus-5-5).
115
+ *
116
+ * Source: the Messages API thinking table (cached 2026-09-25). Opus 5 'off' is low effort rather
117
+ * than `disabled`, which Anthropic documents as leaking tool calls and thinking tags into text.
118
+ */
119
+ const ANTHROPIC_FAMILIES = Object.freeze([
120
+ { family: 'claude-fable / claude-opus-5-5', match: /fable|mythos|opus-5-5/,
121
+ examples: ['claude-opus-5-5', 'claude-fable-5-1', 'claude-fable-5', 'claude-mythos-5-1'],
122
+ off: { effort: 'low' }, full: { effort: 'high' }, sampling: false },
123
+ { family: 'claude-opus-5', match: /opus-5(?![-.]?\d)/,
124
+ examples: ['claude-opus-5'],
125
+ off: { effort: 'low' }, full: { thinking: { type: 'adaptive' }, effort: 'high' }, sampling: false },
126
+ { family: 'claude-sonnet-5-5', match: /sonnet-5-5/,
127
+ examples: ['claude-sonnet-5-5'],
128
+ off: { thinking: { type: 'between_tools' } }, full: { thinking: { type: 'adaptive' }, effort: 'high' }, sampling: false },
129
+ { family: 'claude-sonnet-5', match: /sonnet-5(?![-.]?\d)/,
130
+ examples: ['claude-sonnet-5'],
131
+ off: { thinking: { type: 'disabled' } }, full: { thinking: { type: 'adaptive' }, effort: 'high' }, sampling: false },
132
+ { family: 'claude-opus-4-7 / 4-8', match: /opus-4-[78]/,
133
+ examples: ['claude-opus-4-8', 'claude-opus-4-7'],
134
+ off: {}, full: { thinking: { type: 'adaptive' }, effort: 'high' }, sampling: false },
135
+ { family: 'claude-*-4-6', match: /(opus|sonnet)-4-6/,
136
+ examples: ['claude-opus-4-6', 'claude-sonnet-4-6'],
137
+ off: {}, full: { thinking: { type: 'adaptive' }, effort: 'high' }, sampling: true },
138
+ { family: 'claude-*-4-5 and older', match: /-4-5|-4-1|(opus|sonnet|haiku)-4(?![-.]?\d)|claude-3/,
139
+ examples: ['claude-haiku-4-5', 'claude-sonnet-4-5', 'claude-opus-4-5', 'claude-opus-4-1', 'claude-3-7-sonnet-latest'],
140
+ off: {}, full: { budget: true }, sampling: true }
141
+ ]);
142
+
143
+ /** Thinking room added on top of the answer's own maxTokens when a pre-4.6 model is asked to think. */
144
+ const MIN_THINKING_BUDGET = 1024;
145
+
146
+ /**
147
+ * What to change on an Anthropic request body for this model and mode, or null for an unknown model
148
+ * (which keeps the pre-0.27 request: no thinking field, temperature as given).
149
+ *
150
+ * @returns {{family:string, thinking?:object, effort?:string, sendTemperature:boolean, maxTokens:number}|null}
151
+ */
152
+ function anthropicReasoning(model, mode, maxTokens) {
153
+ assertReasoning(mode);
154
+ const m = String(model || '').toLowerCase();
155
+ const f = ANTHROPIC_FAMILIES.find((x) => x.match.test(m));
156
+ if (!f) return null;
157
+ const want = isFull(mode) ? f.full : f.off;
158
+ const out = { family: f.family, sendTemperature: f.sampling, maxTokens };
159
+ if (want.thinking) out.thinking = want.thinking;
160
+ if (want.effort) out.effort = want.effort;
161
+ if (want.budget) {
162
+ // The caller's maxTokens stays the ANSWER's room; the thinking budget is added on top, because
163
+ // budget_tokens must be below max_tokens and a budget carved out of the answer would starve it.
164
+ const budget = Math.max(MIN_THINKING_BUDGET, maxTokens);
165
+ out.thinking = { type: 'enabled', budget_tokens: budget };
166
+ out.maxTokens = maxTokens + budget;
167
+ // Extended thinking requires the default temperature on these models.
168
+ out.sendTemperature = false;
169
+ }
170
+ return out;
171
+ }
172
+
173
+ module.exports = {
174
+ REASONING_MODES, REASONING_FAMILIES, LMSTUDIO_EXTRA, LMSTUDIO_PROMPT_SWITCH, ANTHROPIC_FAMILIES, MIN_THINKING_BUDGET,
175
+ assertReasoning, reasoningFor, lmstudioPromptSwitch, anthropicReasoning
176
+ };