@aria-framework/ai 0.24.1 → 0.26.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lmxStatus.js CHANGED
@@ -50,7 +50,7 @@ function reasoningLabel(model) {
50
50
  const flag = reasoningFor(model);
51
51
  if (!flag) return null;
52
52
  if (flag.reasoning_effort) return `reasoning_effort: ${flag.reasoning_effort}`;
53
- return 'enable_thinking: false';
53
+ return `enable_thinking: ${flag.chat_template_kwargs.enable_thinking}`;
54
54
  }
55
55
 
56
56
  /**
@@ -155,12 +155,16 @@ const rowRef = (row) => ({ id: row.lmx_engine_id || null, name: row.lmx_engine }
155
155
  */
156
156
  function rekeyPlan(rows, engines) {
157
157
  if (!Array.isArray(engines)) return [];
158
+ // IDS ROWS ALREADY TRACK (0.25.0). CLIENT.md's condition for minting is "an id you have never
159
+ // seen". Without it, renaming row A's engine to the name a stale, id-less row B still carries
160
+ // handed B A's id — B's routes silently went to A's engine.
161
+ const held = new Set((rows || []).map((r) => r && r.lmx_engine_id).filter(Boolean));
158
162
  const plan = [];
159
163
  for (const row of rows || []) {
160
164
  if (!row) continue;
161
165
  if (!row.lmx_engine_id) {
162
166
  const e = findEngine(engines, { name: row.lmx_engine });
163
- if (e && e.id) plan.push({ id: row.id, set: { lmx_engine_id: e.id }, why: 'minted' });
167
+ if (e && e.id && !held.has(e.id)) plan.push({ id: row.id, set: { lmx_engine_id: e.id }, why: 'minted' });
164
168
  continue;
165
169
  }
166
170
  const e = findEngine(engines, { id: row.lmx_engine_id });
package/lmxVerify.js CHANGED
@@ -216,6 +216,11 @@ async function verify(o = {}) {
216
216
  if (doc && doc.instance && String(doc.instance) !== instance) {
217
217
  checks.push(fail('instance',
218
218
  `this stack reports itself as “${doc.instance}”, not “${instance}” — the address points at a different deployment`));
219
+ // AND ITS ENGINES ARE NOT HANDED BACK (0.25.0). The apps' scheduled refresh feeds
220
+ // report.engines to rekeyPlan; returning another deployment's list wrote THAT stack's engine
221
+ // ids onto this stack's rows — irreversibly, since a row with an id is never re-pointed.
222
+ // `null` is "no document for this stack", which every consumer treats as "write nothing".
223
+ found = null;
219
224
  }
220
225
 
221
226
  checks.push(await enginesKeyCheck(o, engines, timeoutMs));
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@aria-framework/ai",
3
3
  "description": "Aria App Framework \u2014 AI module. A dependency-injected model seam (createAiClient) over several providers (LM Studio / OpenAI-compatible / Anthropic), with a fact-preservation guard, generic Polish and Generate writing engines, and a browser polish widget. Prompts and config stay in the consuming app.",
4
- "version": "0.24.1",
4
+ "version": "0.26.0",
5
5
  "license": "UNLICENSED",
6
6
  "private": false,
7
7
  "publishConfig": {
@@ -47,7 +47,7 @@
47
47
  }
48
48
  },
49
49
  "scripts": {
50
- "test": "node test/smoke.js && node test/usageStore.js && node test/providerStore.js && node test/speedStore.js && node test/health.js && node test/listModels.js && node test/benchmark.js && node test/lmxDiscovery.js && node test/lmx.js && node test/lmxVerify.js && node test/lmxStore.js && node test/lmxStatus.js && node test/jobCard.js && node test/untrusted.js && node test/packaging.js && node test/views.js && node test/fenced.js && node test/polish.js",
50
+ "test": "node test/smoke.js && node test/usageStore.js && node test/providerStore.js && node test/speedStore.js && node test/health.js && node test/listModels.js && node test/benchmark.js && node test/lmxDiscovery.js && node test/lmx.js && node test/lmxVerify.js && node test/lmxStore.js && node test/lmxStatus.js && node test/jobCard.js && node test/untrusted.js && node test/packaging.js && node test/views.js && node test/fenced.js && node test/polish.js && node test/reasoning.js",
51
51
  "prepublishOnly": "node ../../test/packaging.js ai"
52
52
  },
53
53
  "devDependencies": {
package/providers/lmx.js CHANGED
@@ -27,16 +27,33 @@ const { createLmxDiscovery } = require('./lmxDiscovery');
27
27
  const { lmxTransport } = require('./lmxTransport');
28
28
  const { AiError } = require('../error');
29
29
 
30
- /** Discovery instances, keyed so several engines on one stack share a single poller. */
30
+ /**
31
+ * Discovery instances, ONE PER STACK (keyed by instance), so several engines on one stack share a
32
+ * single poller.
33
+ *
34
+ * Keyed by instance alone since 0.25.0. Keyed by (instance, statusUrl), an edited status URL created
35
+ * a SECOND poller and left the first polling the old address every 2 s for the life of the process.
36
+ */
31
37
  const registry = new Map();
32
38
 
33
- const keyOf = (lmx) => `${lmx.instance}\u0000${lmx.statusUrl}`;
39
+ /** The settings a live poller takes from the app's supervisor row, applied on every use. */
40
+ const settingsOf = (lmx) => ({
41
+ token: lmx.statusToken,
42
+ ca: lmx.ca || null,
43
+ staleMs: lmx.staleMs,
44
+ pollMs: lmx.pollMs,
45
+ fetchTimeoutMs: lmx.statusTimeoutMs
46
+ });
34
47
 
35
48
  /**
36
- * The discovery for this supervisor, created once and started on first use.
49
+ * The discovery for this supervisor, created once and started on first use — and RECONFIGURED on
50
+ * every use (0.25.0).
37
51
  *
38
- * Credentials are NOT part of the key: rotating a token should not orphan a running poller and
39
- * leave the old one polling with a dead credential. The live instance is updated in place instead.
52
+ * Credentials are not part of the key, so rotating a token does not orphan the poller; until 0.25.0
53
+ * nothing then applied the new token either (the comment here said "updated in place"; nothing did),
54
+ * so a rotated key or a replaced certificate never reached the poller and every lmx endpoint went
55
+ * stale until a restart. configure() applies whatever the app hands over each time — the app
56
+ * resolves its supervisor row per call, so the row is the source of truth.
40
57
  */
41
58
  function discoveryFor(lmx, logger) {
42
59
  if (!lmx || !lmx.instance || !lmx.statusUrl) {
@@ -44,29 +61,44 @@ function discoveryFor(lmx, logger) {
44
61
  'This endpoint is an lmx engine but no supervisor is configured for it — an engine name '
45
62
  + 'without a status listener cannot be resolved to an address.');
46
63
  }
47
- const key = keyOf(lmx);
64
+ const key = lmx.instance;
48
65
  let d = registry.get(key);
49
- if (!d) {
50
- d = createLmxDiscovery({
66
+ if (d && d.status().statusUrl !== lmx.statusUrl) {
67
+ // A NEW ADDRESS IS A NEW POLLER: the document, its age and its last error belong to the old one.
68
+ d.stop();
69
+ registry.delete(key);
70
+ d = null;
71
+ }
72
+ if (d) {
73
+ d.configure(settingsOf(lmx));
74
+ } else {
75
+ d = createLmxDiscovery(Object.assign({
51
76
  instance: lmx.instance,
52
77
  statusUrl: lmx.statusUrl,
53
- token: lmx.statusToken,
54
- ca: lmx.ca || null,
55
- staleMs: lmx.staleMs,
56
- // Was accepted from the app's supervisor row and then dropped here, so a configured poll
57
- // interval never took effect. undefined falls through to POLL_MS.
58
- pollMs: lmx.pollMs,
59
78
  // Test seam, the same one createLmxDiscovery already takes: it is the only way to exercise
60
79
  // complete() end-to-end without a live stack, and the cold-start bug lived exactly there.
61
80
  fetchImpl: lmx.fetchImpl,
62
81
  logger
63
- });
82
+ }, settingsOf(lmx)));
64
83
  d.start();
65
84
  registry.set(key, d);
66
85
  }
67
86
  return d;
68
87
  }
69
88
 
89
+ /**
90
+ * Stop and drop the poller for a stack — call it when the app deletes or disables a supervisor
91
+ * (0.25.0). Otherwise that stack's poller kept polling every 2 s for the life of the process.
92
+ * @returns {boolean} whether there was one to forget
93
+ */
94
+ function forgetDiscovery(instance) {
95
+ const d = registry.get(instance);
96
+ if (!d) return false;
97
+ d.stop();
98
+ registry.delete(instance);
99
+ return true;
100
+ }
101
+
70
102
  /**
71
103
  * Why a resolution failed, in the vocabulary the dispatcher classifies on.
72
104
  *
@@ -117,12 +149,43 @@ function skipError(name, res) {
117
149
  * An UNRECOGNISED family gets NEITHER flag. Guessing would be worse than not guessing — sending a
118
150
  * Qwen argument to a model that ignores it wastes the budget silently, which is the failure this
119
151
  * exists to avoid.
152
+ *
153
+ * THE DEFAULT IS "THINK AS LITTLE AS THE FAMILY ALLOWS"; `reasoning: 'full'` (0.26.0) is the one
154
+ * named way to ask for the opposite — Qwen thinking on, gpt-oss effort high — for judgement work
155
+ * (triage) where a thinking-off answer measured as a different product. It is a NAMED per-call
156
+ * option rather than a raw `extra`, because the forced flag deliberately outranks `extra`: a stray
157
+ * passthrough must never be able to turn thinking on and spend a budget sized for a quick answer.
120
158
  */
121
- function reasoningFor(modelPath) {
159
+ const REASONING_MODES = Object.freeze(['full']);
160
+
161
+ /**
162
+ * Refuse a `reasoning` value this package does not define. `undefined`/`null` mean the default.
163
+ * Anything else — 'high', 'Full', true, '' — is a caller's mistake, and silently treating it as the
164
+ * default would hand a triage caller a thinking-off answer while it believed it had asked for more.
165
+ */
166
+ function assertReasoning(value) {
167
+ if (value === undefined || value === null) return;
168
+ if (!REASONING_MODES.includes(value)) {
169
+ throw new AiError('refused',
170
+ `Unknown reasoning option ${JSON.stringify(value)}. The only value is `
171
+ + `${REASONING_MODES.map((m) => `'${m}'`).join(', ')}; leave it out for the default.`);
172
+ }
173
+ }
174
+
175
+ /**
176
+ * The families with a known flag, in match order. A TABLE rather than a chain of ifs so the README
177
+ * test can enumerate it: a family added here and not documented fails that test.
178
+ */
179
+ const REASONING_FAMILIES = Object.freeze([
180
+ { family: 'qwen', match: /qwen/, flag: (full) => ({ chat_template_kwargs: { enable_thinking: full } }) },
181
+ { family: 'gpt-oss', match: /gpt-oss/, flag: (full) => ({ reasoning_effort: full ? 'high' : 'low' }) }
182
+ ]);
183
+
184
+ function reasoningFor(modelPath, mode) {
185
+ assertReasoning(mode);
122
186
  const m = String(modelPath || '').toLowerCase();
123
- if (/qwen/.test(m)) return { chat_template_kwargs: { enable_thinking: false } };
124
- if (/gpt-oss/.test(m)) return { reasoning_effort: 'low' };
125
- return null;
187
+ const f = REASONING_FAMILIES.find((x) => x.match.test(m));
188
+ return f ? f.flag(mode === 'full') : null;
126
189
  }
127
190
 
128
191
  /** The config openai-compatible needs, once the address is known. */
@@ -172,23 +235,31 @@ function tagThrottle(err) {
172
235
  }
173
236
 
174
237
  async function complete(cfg, opts) {
238
+ // Validated BEFORE discovery, so a typo is refused without a status read or an engine call.
239
+ const mode = opts && opts.reasoning;
240
+ assertReasoning(mode);
175
241
  const d = discoveryFor(cfg.lmx, cfg.logger);
176
242
  const name = cfg.lmx.engine;
177
243
  const res = await resolveWarm(d, refOf(cfg.lmx));
178
244
  if (!res.url) throw skipError(name, res);
179
245
 
180
- const reasoning = reasoningFor(res.engine.model);
246
+ const reasoning = reasoningFor(res.engine.model, mode);
181
247
  if (!reasoning && cfg.logger) {
182
- cfg.logger.warn(
183
- `lmx: no reasoning flag known for model "${res.engine.model}" on engine "${name}" — sending `
184
- + 'none. If it reasons before answering, the token budget may be consumed with no content '
185
- + 'returned.'
248
+ cfg.logger.warn(mode === 'full'
249
+ ? `lmx: reasoning '${mode}' was asked for, but no reasoning flag is known for model `
250
+ + `"${res.engine.model}" on engine "${name}" — sending none; the model runs on its own default.`
251
+ : `lmx: no reasoning flag known for model "${res.engine.model}" on engine "${name}" — sending `
252
+ + 'none. If it reasons before answering, the token budget may be consumed with no content '
253
+ + 'returned.'
186
254
  );
187
255
  }
188
256
 
257
+ // The named option is consumed here; openai-compatible never sees it.
258
+ const { reasoning: _consumed, ...rest } = opts || {};
189
259
  return openai.complete(engineConfig(cfg, res.url, res.engine), {
190
- ...opts,
191
- // Merged rather than replacing: a caller's own extras survive.
260
+ ...rest,
261
+ // Merged rather than replacing: a caller's own extras survive. The flag is spread LAST, so a
262
+ // raw extra can never override it — only the named `reasoning` option changes what it says.
192
263
  extra: { ...(opts && opts.extra), ...(reasoning || {}) }
193
264
  }).catch((err) => { throw tagThrottle(err); });
194
265
  }
@@ -257,5 +328,6 @@ function _resetRegistry() {
257
328
  module.exports = {
258
329
  complete, embed, listModels, listModelsResult,
259
330
  apiRoot: openai.apiRoot,
260
- reasoningFor, discoveryFor, _resetRegistry, SKIP_CODE
331
+ reasoningFor, assertReasoning, REASONING_MODES, REASONING_FAMILIES,
332
+ discoveryFor, forgetDiscovery, _resetRegistry, SKIP_CODE
261
333
  };
@@ -56,6 +56,25 @@ const STALE_MS = 3 * 60 * 1000;
56
56
  /** The only state that may receive new work. */
57
57
  const HEALTHY = 'healthy';
58
58
 
59
+ /**
60
+ * The fastest a poller may run (0.25.0). A poll interval of 0 — a NULL coerced, a typo — is not a
61
+ * faster poll, it is a busy loop against the supervisor.
62
+ */
63
+ const MIN_POLL_MS = 500;
64
+
65
+ /**
66
+ * How long one status read may take (0.25.0). There was no bound at all — only undici's 300 s
67
+ * defaults — so after a restart an lmx call awaited a hung listener far past its own timeout and
68
+ * the dispatcher never failed over. The document is small and served no-store: seconds is ample.
69
+ */
70
+ const FETCH_TIMEOUT_MS = 5000;
71
+
72
+ // null/undefined mean "the default" (lmxStore stores NULL for exactly that); any number — 0, a
73
+ // typo, NaN — is floored rather than trusted.
74
+ const floorPoll = (ms) => (ms === undefined || ms === null || ms === ''
75
+ ? POLL_MS
76
+ : Math.max(MIN_POLL_MS, Number(ms) || 0));
77
+
59
78
  /**
60
79
  * @param {object} opts
61
80
  * @param {string} opts.statusUrl e.g. https://host:9443/status
@@ -63,7 +82,8 @@ const HEALTHY = 'healthy';
63
82
  * @param {string} opts.token bearer for the status listener
64
83
  * @param {string} [opts.ca] PEM of the certificate to pin
65
84
  * @param {number} [opts.staleMs]
66
- * @param {number} [opts.pollMs]
85
+ * @param {number} [opts.pollMs] floored at MIN_POLL_MS
86
+ * @param {number} [opts.fetchTimeoutMs] one status read's deadline (default FETCH_TIMEOUT_MS)
67
87
  * @param {object} [opts.logger]
68
88
  * @param {function} [opts.fetchImpl] injectable for tests; defaults to global fetch
69
89
  * @param {function} [opts.now] injectable clock
@@ -72,10 +92,6 @@ function createLmxDiscovery(opts = {}) {
72
92
  const {
73
93
  statusUrl,
74
94
  instance,
75
- token,
76
- ca = null,
77
- staleMs = STALE_MS,
78
- pollMs = POLL_MS,
79
95
  logger = console,
80
96
  fetchImpl,
81
97
  now = () => Date.now()
@@ -84,10 +100,21 @@ function createLmxDiscovery(opts = {}) {
84
100
  if (!statusUrl) throw new Error('createLmxDiscovery: statusUrl is required');
85
101
  if (!instance) throw new Error('createLmxDiscovery: instance is required — identity is (instance, name)');
86
102
 
103
+ // MUTABLE, AND THAT IS THE 0.25.0 FIX. These were captured once for the life of the process, so a
104
+ // rotated token or a replaced certificate never reached the poller: Check (a fresh fetch) said
105
+ // "verified" while every poll got 401 and, after staleMs, every engine was skipped as stale until
106
+ // a restart. configure() below updates them in place; the registry calls it on every use.
107
+ let token = opts.token;
108
+ let ca = opts.ca || null;
109
+ let staleMs = opts.staleMs || STALE_MS;
110
+ let pollMs = floorPoll(opts.pollMs);
111
+ let fetchTimeoutMs = Number(opts.fetchTimeoutMs) || FETCH_TIMEOUT_MS;
112
+
87
113
  let doc = null; // the last document that parsed and matched our instance
88
114
  let docAt = 0;
89
115
  let lastError = null;
90
116
  let timer = null;
117
+ let inFlight = null; // the one refresh running now; concurrent callers share it
91
118
 
92
119
  /**
93
120
  * The pinned transport — see providers/lmxTransport.js for why this is not NODE_EXTRA_CA_CERTS
@@ -102,7 +129,9 @@ function createLmxDiscovery(opts = {}) {
102
129
  const t = transport();
103
130
  const res = await t.fetch(statusUrl, {
104
131
  headers: { Authorization: `Bearer ${token}` },
105
- dispatcher: t.dispatcher
132
+ dispatcher: t.dispatcher,
133
+ // BOUNDED (0.25.0): a listener that accepts and never answers is abandoned, not awaited.
134
+ signal: AbortSignal.timeout(fetchTimeoutMs)
106
135
  });
107
136
  if (!res.ok) {
108
137
  const err = new Error(`status listener answered ${res.status}`);
@@ -119,7 +148,17 @@ function createLmxDiscovery(opts = {}) {
119
148
  * failure — the listener answered perfectly — so it is reported as a configuration error, which
120
149
  * is what it is. Adopting it would route `analysis` to another deployment's `analysis`.
121
150
  */
122
- async function refresh() {
151
+ /**
152
+ * SINGLE-FLIGHT (0.25.0). The 2 s poller and every warming AI call each started their own read, so
153
+ * a hung listener stacked a new pending fetch per tick and per request. Concurrent callers now
154
+ * share the one in flight.
155
+ */
156
+ function refresh() {
157
+ if (!inFlight) inFlight = refreshOnce().finally(() => { inFlight = null; });
158
+ return inFlight;
159
+ }
160
+
161
+ async function refreshOnce() {
123
162
  try {
124
163
  const next = await fetchOnce();
125
164
  if (next && next.instance && next.instance !== instance) {
@@ -196,8 +235,28 @@ function createLmxDiscovery(opts = {}) {
196
235
  timer = null;
197
236
  }
198
237
 
238
+ /**
239
+ * Apply new settings to a live poller (0.25.0). `undefined` keeps the current value. A new poll
240
+ * interval restarts a running timer so it takes effect now, not after a process restart.
241
+ * The status URL and instance are identity, not settings: a different URL is a different poller
242
+ * (the registry replaces it — see providers/lmx.js).
243
+ */
244
+ function configure(next = {}) {
245
+ if (next.token !== undefined) token = next.token;
246
+ if (next.ca !== undefined) ca = next.ca || null;
247
+ if (next.staleMs !== undefined) staleMs = next.staleMs || STALE_MS;
248
+ if (next.fetchTimeoutMs !== undefined) fetchTimeoutMs = Number(next.fetchTimeoutMs) || FETCH_TIMEOUT_MS;
249
+ if (next.pollMs !== undefined) {
250
+ const p = floorPoll(next.pollMs);
251
+ if (p !== pollMs) {
252
+ pollMs = p;
253
+ if (timer) { stop(); timer = setInterval(refresh, pollMs); if (timer.unref) timer.unref(); }
254
+ }
255
+ }
256
+ }
257
+
199
258
  return {
200
- start, stop, refresh, resolve, engines, engine,
259
+ start, stop, refresh, resolve, engines, engine, configure,
201
260
  /** The pinned transport, so ENGINE calls reach the same host over the same trust. */
202
261
  transport,
203
262
  /** For a diagnostics panel: what we know and how old it is. */
@@ -208,7 +267,9 @@ function createLmxDiscovery(opts = {}) {
208
267
  engineCount: engines().length,
209
268
  ageSec: doc ? ageSec() : null,
210
269
  stale: doc ? isStale() : true,
211
- lastError
270
+ lastError,
271
+ pollMs,
272
+ running: !!timer
212
273
  })
213
274
  };
214
275
  }
@@ -227,4 +288,4 @@ function findEngine(list, ref) {
227
288
  return all.find((e) => e && e.name === r.name) || null;
228
289
  }
229
290
 
230
- module.exports = { createLmxDiscovery, findEngine, POLL_MS, STALE_MS, HEALTHY };
291
+ module.exports = { createLmxDiscovery, findEngine, POLL_MS, STALE_MS, HEALTHY, MIN_POLL_MS, FETCH_TIMEOUT_MS };
@@ -154,8 +154,17 @@ async function complete(cfg, opts) {
154
154
  // that thinking back separately (`reasoning_content`) or inline, wrapped in <think> tags. Neither
155
155
  // is the answer, and both have to be recognised — otherwise a model that reasoned and then ran
156
156
  // out of room looks identical to a broken server.
157
- const reasoning = message.reasoning_content || message.reasoning || '';
158
- const text = stripThinking(message.content || '');
157
+ let reasoning = message.reasoning_content || message.reasoning || '';
158
+ let text = stripThinking(message.content || '');
159
+
160
+ // A LENGTH STOP INSIDE AN UNCLOSED <think> IS NO ANSWER (0.26.0). stripThinking only removes a
161
+ // CLOSED block, so a model that ran out of room mid-thought used to resolve with its half-finished
162
+ // reasoning as the "answer" (truncated: true). Narrow on purpose: only a reply that STARTS with
163
+ // the tag, has no closing tag, and stopped on length — the case reasoning: 'full' makes common.
164
+ if (finishReason === 'length' && unclosedThinking(text)) {
165
+ reasoning = reasoning || text;
166
+ text = '';
167
+ }
159
168
 
160
169
  if (!text) {
161
170
  throw emptyCompletion({ label, finishReason, reasoning, maxTokens: body.max_tokens, usage: payload.usage });
@@ -214,6 +223,12 @@ function stripThinking(content) {
214
223
  .trim();
215
224
  }
216
225
 
226
+ /** A reply that opens a <think>/<thinking> block and never closes it. */
227
+ function unclosedThinking(text) {
228
+ const m = /^<(think|thinking)>/i.exec(String(text || '').trimStart());
229
+ return !!m && !new RegExp(`</${m[1]}>`, 'i').test(text);
230
+ }
231
+
217
232
  /**
218
233
  * Why nothing came back — the message an operator can act on.
219
234
  *
@@ -227,7 +242,8 @@ function emptyCompletion({ label, finishReason, reasoning, maxTokens, usage }) {
227
242
  return new AiError('bad_response',
228
243
  `${label} ran out of room before it answered — it used all ${maxTokens} reply tokens` +
229
244
  (reasoning ? ' on internal reasoning' : '') +
230
- '. This model thinks before it replies, so raise the reply limit (1024 is a sensible floor).',
245
+ '. This model thinks before it replies, so raise the reply limit (1024 is a sensible floor; ' +
246
+ "4096 or more with reasoning: 'full').",
231
247
  { usage });
232
248
  }
233
249
  if (reasoning) {