@aria-framework/ai 0.25.0 → 0.26.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@aria-framework/ai",
3
3
  "description": "Aria App Framework \u2014 AI module. A dependency-injected model seam (createAiClient) over several providers (LM Studio / OpenAI-compatible / Anthropic), with a fact-preservation guard, generic Polish and Generate writing engines, and a browser polish widget. Prompts and config stay in the consuming app.",
4
- "version": "0.25.0",
4
+ "version": "0.26.1",
5
5
  "license": "UNLICENSED",
6
6
  "private": false,
7
7
  "publishConfig": {
@@ -47,7 +47,7 @@
47
47
  }
48
48
  },
49
49
  "scripts": {
50
- "test": "node test/smoke.js && node test/usageStore.js && node test/providerStore.js && node test/speedStore.js && node test/health.js && node test/listModels.js && node test/benchmark.js && node test/lmxDiscovery.js && node test/lmx.js && node test/lmxVerify.js && node test/lmxStore.js && node test/lmxStatus.js && node test/jobCard.js && node test/untrusted.js && node test/packaging.js && node test/views.js && node test/fenced.js && node test/polish.js",
50
+ "test": "node test/smoke.js && node test/usageStore.js && node test/providerStore.js && node test/speedStore.js && node test/health.js && node test/listModels.js && node test/benchmark.js && node test/lmxDiscovery.js && node test/lmx.js && node test/lmxVerify.js && node test/lmxStore.js && node test/lmxStatus.js && node test/jobCard.js && node test/untrusted.js && node test/packaging.js && node test/views.js && node test/fenced.js && node test/polish.js && node test/reasoning.js && node test/usageOnFailure.js",
51
51
  "prepublishOnly": "node ../../test/packaging.js ai"
52
52
  },
53
53
  "devDependencies": {
@@ -97,6 +97,13 @@ async function complete(cfg, opts) {
97
97
  clearTimeout(timer);
98
98
  }
99
99
 
100
+ // A 200 WHOSE BODY IS `null` (0.26.1) is a typed failure, not a TypeError further down.
101
+ if (!payload || typeof payload !== 'object') {
102
+ throw new AiError('bad_response', `${label} answered with an empty body.`);
103
+ }
104
+ // Spent tokens ride on every later error so the client can meter a failed call (0.26.1).
105
+ const usage = normaliseUsage(payload.usage);
106
+
100
107
  // Difference 4: blocks, not a single string. Concatenated so a model that splits its answer over
101
108
  // two text blocks does not silently lose the second one.
102
109
  // Thinking arrives as its own block TYPE here rather than a field, so filtering to 'text' already
@@ -109,7 +116,7 @@ async function complete(cfg, opts) {
109
116
  const thought = (payload.content || []).some((b) => b && b.type === 'thinking');
110
117
  throw new AiError('bad_response', thought
111
118
  ? `${label} returned only its internal reasoning. Raise the reply limit and try again.`
112
- : `${label} returned no text (stop reason: ${payload.stop_reason || 'none given'}).`);
119
+ : `${label} returned no text (stop reason: ${payload.stop_reason || 'none given'}).`, { usage });
113
120
  }
114
121
 
115
122
  let json = null;
@@ -121,10 +128,10 @@ async function complete(cfg, opts) {
121
128
  if (payload.stop_reason === 'max_tokens') {
122
129
  throw new AiError('bad_response',
123
130
  `${label} ran out of room mid-answer — the structured reply was cut off before it was ` +
124
- 'finished. A Regenerate usually succeeds.', { cause: err });
131
+ 'finished. A Regenerate usually succeeds.', { cause: err, usage });
125
132
  }
126
133
  throw new AiError('bad_response',
127
- `${label} was asked for structured output and returned text that will not parse.`, { cause: err });
134
+ `${label} was asked for structured output and returned text that will not parse.`, { cause: err, usage });
128
135
  }
129
136
  }
130
137
 
@@ -132,7 +139,7 @@ async function complete(cfg, opts) {
132
139
  text,
133
140
  json,
134
141
  model: payload.model || cfg.model,
135
- usage: normaliseUsage(payload.usage),
142
+ usage,
136
143
  ms: Date.now() - started,
137
144
  finishReason: payload.stop_reason || null,
138
145
  truncated: payload.stop_reason === 'max_tokens',
package/providers/lmx.js CHANGED
@@ -149,12 +149,43 @@ function skipError(name, res) {
149
149
  * An UNRECOGNISED family gets NEITHER flag. Guessing would be worse than not guessing — sending a
150
150
  * Qwen argument to a model that ignores it wastes the budget silently, which is the failure this
151
151
  * exists to avoid.
152
+ *
153
+ * THE DEFAULT IS "THINK AS LITTLE AS THE FAMILY ALLOWS"; `reasoning: 'full'` (0.26.0) is the one
154
+ * named way to ask for the opposite — Qwen thinking on, gpt-oss effort high — for judgement work
155
+ * (triage) where a thinking-off answer measured as a different product. It is a NAMED per-call
156
+ * option rather than a raw `extra`, because the forced flag deliberately outranks `extra`: a stray
157
+ * passthrough must never be able to turn thinking on and spend a budget sized for a quick answer.
158
+ */
159
+ const REASONING_MODES = Object.freeze(['full']);
160
+
161
+ /**
162
+ * Refuse a `reasoning` value this package does not define. `undefined`/`null` mean the default.
163
+ * Anything else — 'high', 'Full', true, '' — is a caller's mistake, and silently treating it as the
164
+ * default would hand a triage caller a thinking-off answer while it believed it had asked for more.
165
+ */
166
+ function assertReasoning(value) {
167
+ if (value === undefined || value === null) return;
168
+ if (!REASONING_MODES.includes(value)) {
169
+ throw new AiError('refused',
170
+ `Unknown reasoning option ${JSON.stringify(value)}. The only value is `
171
+ + `${REASONING_MODES.map((m) => `'${m}'`).join(', ')}; leave it out for the default.`);
172
+ }
173
+ }
174
+
175
+ /**
176
+ * The families with a known flag, in match order. A TABLE rather than a chain of ifs so the README
177
+ * test can enumerate it: a family added here and not documented fails that test.
152
178
  */
153
- function reasoningFor(modelPath) {
179
+ const REASONING_FAMILIES = Object.freeze([
180
+ { family: 'qwen', match: /qwen/, flag: (full) => ({ chat_template_kwargs: { enable_thinking: full } }) },
181
+ { family: 'gpt-oss', match: /gpt-oss/, flag: (full) => ({ reasoning_effort: full ? 'high' : 'low' }) }
182
+ ]);
183
+
184
+ function reasoningFor(modelPath, mode) {
185
+ assertReasoning(mode);
154
186
  const m = String(modelPath || '').toLowerCase();
155
- if (/qwen/.test(m)) return { chat_template_kwargs: { enable_thinking: false } };
156
- if (/gpt-oss/.test(m)) return { reasoning_effort: 'low' };
157
- return null;
187
+ const f = REASONING_FAMILIES.find((x) => x.match.test(m));
188
+ return f ? f.flag(mode === 'full') : null;
158
189
  }
159
190
 
160
191
  /** The config openai-compatible needs, once the address is known. */
@@ -204,23 +235,31 @@ function tagThrottle(err) {
204
235
  }
205
236
 
206
237
  async function complete(cfg, opts) {
238
+ // Validated BEFORE discovery, so a typo is refused without a status read or an engine call.
239
+ const mode = opts && opts.reasoning;
240
+ assertReasoning(mode);
207
241
  const d = discoveryFor(cfg.lmx, cfg.logger);
208
242
  const name = cfg.lmx.engine;
209
243
  const res = await resolveWarm(d, refOf(cfg.lmx));
210
244
  if (!res.url) throw skipError(name, res);
211
245
 
212
- const reasoning = reasoningFor(res.engine.model);
246
+ const reasoning = reasoningFor(res.engine.model, mode);
213
247
  if (!reasoning && cfg.logger) {
214
- cfg.logger.warn(
215
- `lmx: no reasoning flag known for model "${res.engine.model}" on engine "${name}" — sending `
216
- + 'none. If it reasons before answering, the token budget may be consumed with no content '
217
- + 'returned.'
248
+ cfg.logger.warn(mode === 'full'
249
+ ? `lmx: reasoning '${mode}' was asked for, but no reasoning flag is known for model `
250
+ + `"${res.engine.model}" on engine "${name}" — sending none; the model runs on its own default.`
251
+ : `lmx: no reasoning flag known for model "${res.engine.model}" on engine "${name}" — sending `
252
+ + 'none. If it reasons before answering, the token budget may be consumed with no content '
253
+ + 'returned.'
218
254
  );
219
255
  }
220
256
 
257
+ // The named option is consumed here; openai-compatible never sees it.
258
+ const { reasoning: _consumed, ...rest } = opts || {};
221
259
  return openai.complete(engineConfig(cfg, res.url, res.engine), {
222
- ...opts,
223
- // Merged rather than replacing: a caller's own extras survive.
260
+ ...rest,
261
+ // Merged rather than replacing: a caller's own extras survive. The flag is spread LAST, so a
262
+ // raw extra can never override it — only the named `reasoning` option changes what it says.
224
263
  extra: { ...(opts && opts.extra), ...(reasoning || {}) }
225
264
  }).catch((err) => { throw tagThrottle(err); });
226
265
  }
@@ -289,5 +328,6 @@ function _resetRegistry() {
289
328
  module.exports = {
290
329
  complete, embed, listModels, listModelsResult,
291
330
  apiRoot: openai.apiRoot,
292
- reasoningFor, discoveryFor, forgetDiscovery, _resetRegistry, SKIP_CODE
331
+ reasoningFor, assertReasoning, REASONING_MODES, REASONING_FAMILIES,
332
+ discoveryFor, forgetDiscovery, _resetRegistry, SKIP_CODE
293
333
  };
@@ -133,7 +133,13 @@ async function complete(cfg, opts) {
133
133
  // completion" — which sent this investigation after a thinking-model theory that was not the
134
134
  // cause. Verified against the real server: the same request on /v1/chat/completions answers
135
135
  // normally.
136
- if (payload && payload.error && !payload.choices) {
136
+ // A 200 WHOSE BODY IS `null` (0.26.1) — valid JSON, no object. It used to reach `payload.usage`
137
+ // below as a TypeError, which a router reads as an unknown fault.
138
+ if (!payload || typeof payload !== 'object') {
139
+ throw new AiError('bad_response', `${label} answered with an empty body.`);
140
+ }
141
+
142
+ if (payload.error && !payload.choices) {
137
143
  const said = typeof payload.error === 'string'
138
144
  ? payload.error
139
145
  : (payload.error.message || JSON.stringify(payload.error));
@@ -154,13 +160,26 @@ async function complete(cfg, opts) {
154
160
  // that thinking back separately (`reasoning_content`) or inline, wrapped in <think> tags. Neither
155
161
  // is the answer, and both have to be recognised — otherwise a model that reasoned and then ran
156
162
  // out of room looks identical to a broken server.
157
- const reasoning = message.reasoning_content || message.reasoning || '';
158
- const text = stripThinking(message.content || '');
163
+ let reasoning = message.reasoning_content || message.reasoning || '';
164
+ let text = stripThinking(message.content || '');
165
+
166
+ // A LENGTH STOP INSIDE AN UNCLOSED <think> IS NO ANSWER (0.26.0). stripThinking only removes a
167
+ // CLOSED block, so a model that ran out of room mid-thought used to resolve with its half-finished
168
+ // reasoning as the "answer" (truncated: true). Narrow on purpose: only a reply that STARTS with
169
+ // the tag, has no closing tag, and stopped on length — the case reasoning: 'full' makes common.
170
+ if (finishReason === 'length' && unclosedThinking(text)) {
171
+ reasoning = reasoning || text;
172
+ text = '';
173
+ }
159
174
 
160
175
  if (!text) {
161
176
  throw emptyCompletion({ label, finishReason, reasoning, maxTokens: body.max_tokens, usage: payload.usage });
162
177
  }
163
178
 
179
+ // TOKENS WERE SPENT EVEN WHEN THE ANSWER IS UNUSABLE (0.26.1). Carried on every error raised after
180
+ // this point, normalised, so the client can meter a failed call — see index.js complete().
181
+ const usage = normaliseUsage(payload.usage);
182
+
164
183
  // With a schema the content is still a STRING containing JSON — the server constrains the shape,
165
184
  // it does not parse for you. Parsing here rather than at each call site means a malformed body is
166
185
  // one error type instead of a surprise in a page render.
@@ -179,10 +198,10 @@ async function complete(cfg, opts) {
179
198
  throw new AiError('bad_response',
180
199
  `${label} ran out of room mid-answer (${body.max_tokens} tokens) — the structured reply ` +
181
200
  'was cut off before it was finished. If this repeats, the schema may be letting the ' +
182
- 'model ramble; a Regenerate usually succeeds.', { cause: err });
201
+ 'model ramble; a Regenerate usually succeeds.', { cause: err, usage });
183
202
  }
184
203
  throw new AiError('bad_response',
185
- `${label} was asked for structured output and returned text that will not parse.`, { cause: err });
204
+ `${label} was asked for structured output and returned text that will not parse.`, { cause: err, usage });
186
205
  }
187
206
  }
188
207
 
@@ -190,7 +209,7 @@ async function complete(cfg, opts) {
190
209
  text,
191
210
  json,
192
211
  model: payload.model || cfg.model,
193
- usage: normaliseUsage(payload.usage),
212
+ usage,
194
213
  ms: Date.now() - started,
195
214
  finishReason,
196
215
  // A caller that cares (the summary's coverage line) can tell a complete answer from a cut-off
@@ -214,6 +233,12 @@ function stripThinking(content) {
214
233
  .trim();
215
234
  }
216
235
 
236
+ /** A reply that opens a <think>/<thinking> block and never closes it. */
237
+ function unclosedThinking(text) {
238
+ const m = /^<(think|thinking)>/i.exec(String(text || '').trimStart());
239
+ return !!m && !new RegExp(`</${m[1]}>`, 'i').test(text);
240
+ }
241
+
217
242
  /**
218
243
  * Why nothing came back — the message an operator can act on.
219
244
  *
@@ -221,13 +246,15 @@ function stripThinking(content) {
221
246
  * payload, so they are distinguished: the budget ran out, the budget went on thinking, or the server
222
247
  * genuinely said nothing.
223
248
  */
224
- function emptyCompletion({ label, finishReason, reasoning, maxTokens, usage }) {
225
- const spent = (usage && usage.completion_tokens) || 0;
249
+ function emptyCompletion({ label, finishReason, reasoning, maxTokens, usage: raw }) {
250
+ const spent = (raw && raw.completion_tokens) || 0;
251
+ const usage = normaliseUsage(raw);
226
252
  if (finishReason === 'length' || (reasoning && !spentLeftRoom(spent, maxTokens))) {
227
253
  return new AiError('bad_response',
228
254
  `${label} ran out of room before it answered — it used all ${maxTokens} reply tokens` +
229
255
  (reasoning ? ' on internal reasoning' : '') +
230
- '. This model thinks before it replies, so raise the reply limit (1024 is a sensible floor).',
256
+ '. This model thinks before it replies, so raise the reply limit (1024 is a sensible floor; ' +
257
+ "4096 or more with reasoning: 'full').",
231
258
  { usage });
232
259
  }
233
260
  if (reasoning) {
@@ -237,7 +264,7 @@ function emptyCompletion({ label, finishReason, reasoning, maxTokens, usage }) {
237
264
  }
238
265
  return new AiError('bad_response',
239
266
  `${label} returned an empty completion (finish reason: ${finishReason || 'none given'}). ` +
240
- 'Check the model is fully loaded on the server.');
267
+ 'Check the model is fully loaded on the server.', { usage });
241
268
  }
242
269
 
243
270
  /** Did the completion stop well short of the ceiling? Then the ceiling was not the problem. */