@aria-framework/ai 0.24.1 → 0.26.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +72 -13
- package/index.js +333 -316
- package/lmxStatus.js +6 -2
- package/lmxVerify.js +5 -0
- package/package.json +2 -2
- package/providers/lmx.js +99 -27
- package/providers/lmxDiscovery.js +71 -10
- package/providers/openai-compatible.js +19 -3
package/lmxStatus.js
CHANGED
|
@@ -50,7 +50,7 @@ function reasoningLabel(model) {
|
|
|
50
50
|
const flag = reasoningFor(model);
|
|
51
51
|
if (!flag) return null;
|
|
52
52
|
if (flag.reasoning_effort) return `reasoning_effort: ${flag.reasoning_effort}`;
|
|
53
|
-
return
|
|
53
|
+
return `enable_thinking: ${flag.chat_template_kwargs.enable_thinking}`;
|
|
54
54
|
}
|
|
55
55
|
|
|
56
56
|
/**
|
|
@@ -155,12 +155,16 @@ const rowRef = (row) => ({ id: row.lmx_engine_id || null, name: row.lmx_engine }
|
|
|
155
155
|
*/
|
|
156
156
|
function rekeyPlan(rows, engines) {
|
|
157
157
|
if (!Array.isArray(engines)) return [];
|
|
158
|
+
// IDS ROWS ALREADY TRACK (0.25.0). CLIENT.md's condition for minting is "an id you have never
|
|
159
|
+
// seen". Without it, renaming row A's engine to the name a stale, id-less row B still carries
|
|
160
|
+
// handed B A's id — B's routes silently went to A's engine.
|
|
161
|
+
const held = new Set((rows || []).map((r) => r && r.lmx_engine_id).filter(Boolean));
|
|
158
162
|
const plan = [];
|
|
159
163
|
for (const row of rows || []) {
|
|
160
164
|
if (!row) continue;
|
|
161
165
|
if (!row.lmx_engine_id) {
|
|
162
166
|
const e = findEngine(engines, { name: row.lmx_engine });
|
|
163
|
-
if (e && e.id) plan.push({ id: row.id, set: { lmx_engine_id: e.id }, why: 'minted' });
|
|
167
|
+
if (e && e.id && !held.has(e.id)) plan.push({ id: row.id, set: { lmx_engine_id: e.id }, why: 'minted' });
|
|
164
168
|
continue;
|
|
165
169
|
}
|
|
166
170
|
const e = findEngine(engines, { id: row.lmx_engine_id });
|
package/lmxVerify.js
CHANGED
|
@@ -216,6 +216,11 @@ async function verify(o = {}) {
|
|
|
216
216
|
if (doc && doc.instance && String(doc.instance) !== instance) {
|
|
217
217
|
checks.push(fail('instance',
|
|
218
218
|
`this stack reports itself as “${doc.instance}”, not “${instance}” — the address points at a different deployment`));
|
|
219
|
+
// AND ITS ENGINES ARE NOT HANDED BACK (0.25.0). The apps' scheduled refresh feeds
|
|
220
|
+
// report.engines to rekeyPlan; returning another deployment's list wrote THAT stack's engine
|
|
221
|
+
// ids onto this stack's rows — irreversibly, since a row with an id is never re-pointed.
|
|
222
|
+
// `null` is "no document for this stack", which every consumer treats as "write nothing".
|
|
223
|
+
found = null;
|
|
219
224
|
}
|
|
220
225
|
|
|
221
226
|
checks.push(await enginesKeyCheck(o, engines, timeoutMs));
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@aria-framework/ai",
|
|
3
3
|
"description": "Aria App Framework \u2014 AI module. A dependency-injected model seam (createAiClient) over several providers (LM Studio / OpenAI-compatible / Anthropic), with a fact-preservation guard, generic Polish and Generate writing engines, and a browser polish widget. Prompts and config stay in the consuming app.",
|
|
4
|
-
"version": "0.
|
|
4
|
+
"version": "0.26.0",
|
|
5
5
|
"license": "UNLICENSED",
|
|
6
6
|
"private": false,
|
|
7
7
|
"publishConfig": {
|
|
@@ -47,7 +47,7 @@
|
|
|
47
47
|
}
|
|
48
48
|
},
|
|
49
49
|
"scripts": {
|
|
50
|
-
"test": "node test/smoke.js && node test/usageStore.js && node test/providerStore.js && node test/speedStore.js && node test/health.js && node test/listModels.js && node test/benchmark.js && node test/lmxDiscovery.js && node test/lmx.js && node test/lmxVerify.js && node test/lmxStore.js && node test/lmxStatus.js && node test/jobCard.js && node test/untrusted.js && node test/packaging.js && node test/views.js && node test/fenced.js && node test/polish.js",
|
|
50
|
+
"test": "node test/smoke.js && node test/usageStore.js && node test/providerStore.js && node test/speedStore.js && node test/health.js && node test/listModels.js && node test/benchmark.js && node test/lmxDiscovery.js && node test/lmx.js && node test/lmxVerify.js && node test/lmxStore.js && node test/lmxStatus.js && node test/jobCard.js && node test/untrusted.js && node test/packaging.js && node test/views.js && node test/fenced.js && node test/polish.js && node test/reasoning.js",
|
|
51
51
|
"prepublishOnly": "node ../../test/packaging.js ai"
|
|
52
52
|
},
|
|
53
53
|
"devDependencies": {
|
package/providers/lmx.js
CHANGED
|
@@ -27,16 +27,33 @@ const { createLmxDiscovery } = require('./lmxDiscovery');
|
|
|
27
27
|
const { lmxTransport } = require('./lmxTransport');
|
|
28
28
|
const { AiError } = require('../error');
|
|
29
29
|
|
|
30
|
-
/**
|
|
30
|
+
/**
|
|
31
|
+
* Discovery instances, ONE PER STACK (keyed by instance), so several engines on one stack share a
|
|
32
|
+
* single poller.
|
|
33
|
+
*
|
|
34
|
+
* Keyed by instance alone since 0.25.0. Keyed by (instance, statusUrl), an edited status URL created
|
|
35
|
+
* a SECOND poller and left the first polling the old address every 2 s for the life of the process.
|
|
36
|
+
*/
|
|
31
37
|
const registry = new Map();
|
|
32
38
|
|
|
33
|
-
|
|
39
|
+
/** The settings a live poller takes from the app's supervisor row, applied on every use. */
|
|
40
|
+
const settingsOf = (lmx) => ({
|
|
41
|
+
token: lmx.statusToken,
|
|
42
|
+
ca: lmx.ca || null,
|
|
43
|
+
staleMs: lmx.staleMs,
|
|
44
|
+
pollMs: lmx.pollMs,
|
|
45
|
+
fetchTimeoutMs: lmx.statusTimeoutMs
|
|
46
|
+
});
|
|
34
47
|
|
|
35
48
|
/**
|
|
36
|
-
* The discovery for this supervisor, created once and started on first use
|
|
49
|
+
* The discovery for this supervisor, created once and started on first use — and RECONFIGURED on
|
|
50
|
+
* every use (0.25.0).
|
|
37
51
|
*
|
|
38
|
-
* Credentials are
|
|
39
|
-
*
|
|
52
|
+
* Credentials are not part of the key, so rotating a token does not orphan the poller; until 0.25.0
|
|
53
|
+
* nothing then applied the new token either (the comment here said "updated in place"; nothing did),
|
|
54
|
+
* so a rotated key or a replaced certificate never reached the poller and every lmx endpoint went
|
|
55
|
+
* stale until a restart. configure() applies whatever the app hands over each time — the app
|
|
56
|
+
* resolves its supervisor row per call, so the row is the source of truth.
|
|
40
57
|
*/
|
|
41
58
|
function discoveryFor(lmx, logger) {
|
|
42
59
|
if (!lmx || !lmx.instance || !lmx.statusUrl) {
|
|
@@ -44,29 +61,44 @@ function discoveryFor(lmx, logger) {
|
|
|
44
61
|
'This endpoint is an lmx engine but no supervisor is configured for it — an engine name '
|
|
45
62
|
+ 'without a status listener cannot be resolved to an address.');
|
|
46
63
|
}
|
|
47
|
-
const key =
|
|
64
|
+
const key = lmx.instance;
|
|
48
65
|
let d = registry.get(key);
|
|
49
|
-
if (
|
|
50
|
-
|
|
66
|
+
if (d && d.status().statusUrl !== lmx.statusUrl) {
|
|
67
|
+
// A NEW ADDRESS IS A NEW POLLER: the document, its age and its last error belong to the old one.
|
|
68
|
+
d.stop();
|
|
69
|
+
registry.delete(key);
|
|
70
|
+
d = null;
|
|
71
|
+
}
|
|
72
|
+
if (d) {
|
|
73
|
+
d.configure(settingsOf(lmx));
|
|
74
|
+
} else {
|
|
75
|
+
d = createLmxDiscovery(Object.assign({
|
|
51
76
|
instance: lmx.instance,
|
|
52
77
|
statusUrl: lmx.statusUrl,
|
|
53
|
-
token: lmx.statusToken,
|
|
54
|
-
ca: lmx.ca || null,
|
|
55
|
-
staleMs: lmx.staleMs,
|
|
56
|
-
// Was accepted from the app's supervisor row and then dropped here, so a configured poll
|
|
57
|
-
// interval never took effect. undefined falls through to POLL_MS.
|
|
58
|
-
pollMs: lmx.pollMs,
|
|
59
78
|
// Test seam, the same one createLmxDiscovery already takes: it is the only way to exercise
|
|
60
79
|
// complete() end-to-end without a live stack, and the cold-start bug lived exactly there.
|
|
61
80
|
fetchImpl: lmx.fetchImpl,
|
|
62
81
|
logger
|
|
63
|
-
});
|
|
82
|
+
}, settingsOf(lmx)));
|
|
64
83
|
d.start();
|
|
65
84
|
registry.set(key, d);
|
|
66
85
|
}
|
|
67
86
|
return d;
|
|
68
87
|
}
|
|
69
88
|
|
|
89
|
+
/**
|
|
90
|
+
* Stop and drop the poller for a stack — call it when the app deletes or disables a supervisor
|
|
91
|
+
* (0.25.0). Otherwise that stack's poller kept polling every 2 s for the life of the process.
|
|
92
|
+
* @returns {boolean} whether there was one to forget
|
|
93
|
+
*/
|
|
94
|
+
function forgetDiscovery(instance) {
|
|
95
|
+
const d = registry.get(instance);
|
|
96
|
+
if (!d) return false;
|
|
97
|
+
d.stop();
|
|
98
|
+
registry.delete(instance);
|
|
99
|
+
return true;
|
|
100
|
+
}
|
|
101
|
+
|
|
70
102
|
/**
|
|
71
103
|
* Why a resolution failed, in the vocabulary the dispatcher classifies on.
|
|
72
104
|
*
|
|
@@ -117,12 +149,43 @@ function skipError(name, res) {
|
|
|
117
149
|
* An UNRECOGNISED family gets NEITHER flag. Guessing would be worse than not guessing — sending a
|
|
118
150
|
* Qwen argument to a model that ignores it wastes the budget silently, which is the failure this
|
|
119
151
|
* exists to avoid.
|
|
152
|
+
*
|
|
153
|
+
* THE DEFAULT IS "THINK AS LITTLE AS THE FAMILY ALLOWS"; `reasoning: 'full'` (0.26.0) is the one
|
|
154
|
+
* named way to ask for the opposite — Qwen thinking on, gpt-oss effort high — for judgement work
|
|
155
|
+
* (triage) where a thinking-off answer measured as a different product. It is a NAMED per-call
|
|
156
|
+
* option rather than a raw `extra`, because the forced flag deliberately outranks `extra`: a stray
|
|
157
|
+
* passthrough must never be able to turn thinking on and spend a budget sized for a quick answer.
|
|
120
158
|
*/
|
|
121
|
-
|
|
159
|
+
const REASONING_MODES = Object.freeze(['full']);
|
|
160
|
+
|
|
161
|
+
/**
|
|
162
|
+
* Refuse a `reasoning` value this package does not define. `undefined`/`null` mean the default.
|
|
163
|
+
* Anything else — 'high', 'Full', true, '' — is a caller's mistake, and silently treating it as the
|
|
164
|
+
* default would hand a triage caller a thinking-off answer while it believed it had asked for more.
|
|
165
|
+
*/
|
|
166
|
+
function assertReasoning(value) {
|
|
167
|
+
if (value === undefined || value === null) return;
|
|
168
|
+
if (!REASONING_MODES.includes(value)) {
|
|
169
|
+
throw new AiError('refused',
|
|
170
|
+
`Unknown reasoning option ${JSON.stringify(value)}. The only value is `
|
|
171
|
+
+ `${REASONING_MODES.map((m) => `'${m}'`).join(', ')}; leave it out for the default.`);
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
/**
|
|
176
|
+
* The families with a known flag, in match order. A TABLE rather than a chain of ifs so the README
|
|
177
|
+
* test can enumerate it: a family added here and not documented fails that test.
|
|
178
|
+
*/
|
|
179
|
+
const REASONING_FAMILIES = Object.freeze([
|
|
180
|
+
{ family: 'qwen', match: /qwen/, flag: (full) => ({ chat_template_kwargs: { enable_thinking: full } }) },
|
|
181
|
+
{ family: 'gpt-oss', match: /gpt-oss/, flag: (full) => ({ reasoning_effort: full ? 'high' : 'low' }) }
|
|
182
|
+
]);
|
|
183
|
+
|
|
184
|
+
function reasoningFor(modelPath, mode) {
|
|
185
|
+
assertReasoning(mode);
|
|
122
186
|
const m = String(modelPath || '').toLowerCase();
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
return null;
|
|
187
|
+
const f = REASONING_FAMILIES.find((x) => x.match.test(m));
|
|
188
|
+
return f ? f.flag(mode === 'full') : null;
|
|
126
189
|
}
|
|
127
190
|
|
|
128
191
|
/** The config openai-compatible needs, once the address is known. */
|
|
@@ -172,23 +235,31 @@ function tagThrottle(err) {
|
|
|
172
235
|
}
|
|
173
236
|
|
|
174
237
|
async function complete(cfg, opts) {
|
|
238
|
+
// Validated BEFORE discovery, so a typo is refused without a status read or an engine call.
|
|
239
|
+
const mode = opts && opts.reasoning;
|
|
240
|
+
assertReasoning(mode);
|
|
175
241
|
const d = discoveryFor(cfg.lmx, cfg.logger);
|
|
176
242
|
const name = cfg.lmx.engine;
|
|
177
243
|
const res = await resolveWarm(d, refOf(cfg.lmx));
|
|
178
244
|
if (!res.url) throw skipError(name, res);
|
|
179
245
|
|
|
180
|
-
const reasoning = reasoningFor(res.engine.model);
|
|
246
|
+
const reasoning = reasoningFor(res.engine.model, mode);
|
|
181
247
|
if (!reasoning && cfg.logger) {
|
|
182
|
-
cfg.logger.warn(
|
|
183
|
-
`lmx:
|
|
184
|
-
|
|
185
|
-
|
|
248
|
+
cfg.logger.warn(mode === 'full'
|
|
249
|
+
? `lmx: reasoning '${mode}' was asked for, but no reasoning flag is known for model `
|
|
250
|
+
+ `"${res.engine.model}" on engine "${name}" — sending none; the model runs on its own default.`
|
|
251
|
+
: `lmx: no reasoning flag known for model "${res.engine.model}" on engine "${name}" — sending `
|
|
252
|
+
+ 'none. If it reasons before answering, the token budget may be consumed with no content '
|
|
253
|
+
+ 'returned.'
|
|
186
254
|
);
|
|
187
255
|
}
|
|
188
256
|
|
|
257
|
+
// The named option is consumed here; openai-compatible never sees it.
|
|
258
|
+
const { reasoning: _consumed, ...rest } = opts || {};
|
|
189
259
|
return openai.complete(engineConfig(cfg, res.url, res.engine), {
|
|
190
|
-
...
|
|
191
|
-
// Merged rather than replacing: a caller's own extras survive.
|
|
260
|
+
...rest,
|
|
261
|
+
// Merged rather than replacing: a caller's own extras survive. The flag is spread LAST, so a
|
|
262
|
+
// raw extra can never override it — only the named `reasoning` option changes what it says.
|
|
192
263
|
extra: { ...(opts && opts.extra), ...(reasoning || {}) }
|
|
193
264
|
}).catch((err) => { throw tagThrottle(err); });
|
|
194
265
|
}
|
|
@@ -257,5 +328,6 @@ function _resetRegistry() {
|
|
|
257
328
|
module.exports = {
|
|
258
329
|
complete, embed, listModels, listModelsResult,
|
|
259
330
|
apiRoot: openai.apiRoot,
|
|
260
|
-
reasoningFor,
|
|
331
|
+
reasoningFor, assertReasoning, REASONING_MODES, REASONING_FAMILIES,
|
|
332
|
+
discoveryFor, forgetDiscovery, _resetRegistry, SKIP_CODE
|
|
261
333
|
};
|
|
@@ -56,6 +56,25 @@ const STALE_MS = 3 * 60 * 1000;
|
|
|
56
56
|
/** The only state that may receive new work. */
|
|
57
57
|
const HEALTHY = 'healthy';
|
|
58
58
|
|
|
59
|
+
/**
|
|
60
|
+
* The fastest a poller may run (0.25.0). A poll interval of 0 — a NULL coerced, a typo — is not a
|
|
61
|
+
* faster poll, it is a busy loop against the supervisor.
|
|
62
|
+
*/
|
|
63
|
+
const MIN_POLL_MS = 500;
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* How long one status read may take (0.25.0). There was no bound at all — only undici's 300 s
|
|
67
|
+
* defaults — so after a restart an lmx call awaited a hung listener far past its own timeout and
|
|
68
|
+
* the dispatcher never failed over. The document is small and served no-store: seconds is ample.
|
|
69
|
+
*/
|
|
70
|
+
const FETCH_TIMEOUT_MS = 5000;
|
|
71
|
+
|
|
72
|
+
// null/undefined mean "the default" (lmxStore stores NULL for exactly that); any number — 0, a
|
|
73
|
+
// typo, NaN — is floored rather than trusted.
|
|
74
|
+
const floorPoll = (ms) => (ms === undefined || ms === null || ms === ''
|
|
75
|
+
? POLL_MS
|
|
76
|
+
: Math.max(MIN_POLL_MS, Number(ms) || 0));
|
|
77
|
+
|
|
59
78
|
/**
|
|
60
79
|
* @param {object} opts
|
|
61
80
|
* @param {string} opts.statusUrl e.g. https://host:9443/status
|
|
@@ -63,7 +82,8 @@ const HEALTHY = 'healthy';
|
|
|
63
82
|
* @param {string} opts.token bearer for the status listener
|
|
64
83
|
* @param {string} [opts.ca] PEM of the certificate to pin
|
|
65
84
|
* @param {number} [opts.staleMs]
|
|
66
|
-
* @param {number} [opts.pollMs]
|
|
85
|
+
* @param {number} [opts.pollMs] floored at MIN_POLL_MS
|
|
86
|
+
* @param {number} [opts.fetchTimeoutMs] one status read's deadline (default FETCH_TIMEOUT_MS)
|
|
67
87
|
* @param {object} [opts.logger]
|
|
68
88
|
* @param {function} [opts.fetchImpl] injectable for tests; defaults to global fetch
|
|
69
89
|
* @param {function} [opts.now] injectable clock
|
|
@@ -72,10 +92,6 @@ function createLmxDiscovery(opts = {}) {
|
|
|
72
92
|
const {
|
|
73
93
|
statusUrl,
|
|
74
94
|
instance,
|
|
75
|
-
token,
|
|
76
|
-
ca = null,
|
|
77
|
-
staleMs = STALE_MS,
|
|
78
|
-
pollMs = POLL_MS,
|
|
79
95
|
logger = console,
|
|
80
96
|
fetchImpl,
|
|
81
97
|
now = () => Date.now()
|
|
@@ -84,10 +100,21 @@ function createLmxDiscovery(opts = {}) {
|
|
|
84
100
|
if (!statusUrl) throw new Error('createLmxDiscovery: statusUrl is required');
|
|
85
101
|
if (!instance) throw new Error('createLmxDiscovery: instance is required — identity is (instance, name)');
|
|
86
102
|
|
|
103
|
+
// MUTABLE, AND THAT IS THE 0.25.0 FIX. These were captured once for the life of the process, so a
|
|
104
|
+
// rotated token or a replaced certificate never reached the poller: Check (a fresh fetch) said
|
|
105
|
+
// "verified" while every poll got 401 and, after staleMs, every engine was skipped as stale until
|
|
106
|
+
// a restart. configure() below updates them in place; the registry calls it on every use.
|
|
107
|
+
let token = opts.token;
|
|
108
|
+
let ca = opts.ca || null;
|
|
109
|
+
let staleMs = opts.staleMs || STALE_MS;
|
|
110
|
+
let pollMs = floorPoll(opts.pollMs);
|
|
111
|
+
let fetchTimeoutMs = Number(opts.fetchTimeoutMs) || FETCH_TIMEOUT_MS;
|
|
112
|
+
|
|
87
113
|
let doc = null; // the last document that parsed and matched our instance
|
|
88
114
|
let docAt = 0;
|
|
89
115
|
let lastError = null;
|
|
90
116
|
let timer = null;
|
|
117
|
+
let inFlight = null; // the one refresh running now; concurrent callers share it
|
|
91
118
|
|
|
92
119
|
/**
|
|
93
120
|
* The pinned transport — see providers/lmxTransport.js for why this is not NODE_EXTRA_CA_CERTS
|
|
@@ -102,7 +129,9 @@ function createLmxDiscovery(opts = {}) {
|
|
|
102
129
|
const t = transport();
|
|
103
130
|
const res = await t.fetch(statusUrl, {
|
|
104
131
|
headers: { Authorization: `Bearer ${token}` },
|
|
105
|
-
dispatcher: t.dispatcher
|
|
132
|
+
dispatcher: t.dispatcher,
|
|
133
|
+
// BOUNDED (0.25.0): a listener that accepts and never answers is abandoned, not awaited.
|
|
134
|
+
signal: AbortSignal.timeout(fetchTimeoutMs)
|
|
106
135
|
});
|
|
107
136
|
if (!res.ok) {
|
|
108
137
|
const err = new Error(`status listener answered ${res.status}`);
|
|
@@ -119,7 +148,17 @@ function createLmxDiscovery(opts = {}) {
|
|
|
119
148
|
* failure — the listener answered perfectly — so it is reported as a configuration error, which
|
|
120
149
|
* is what it is. Adopting it would route `analysis` to another deployment's `analysis`.
|
|
121
150
|
*/
|
|
122
|
-
|
|
151
|
+
/**
|
|
152
|
+
* SINGLE-FLIGHT (0.25.0). The 2 s poller and every warming AI call each started their own read, so
|
|
153
|
+
* a hung listener stacked a new pending fetch per tick and per request. Concurrent callers now
|
|
154
|
+
* share the one in flight.
|
|
155
|
+
*/
|
|
156
|
+
function refresh() {
|
|
157
|
+
if (!inFlight) inFlight = refreshOnce().finally(() => { inFlight = null; });
|
|
158
|
+
return inFlight;
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
async function refreshOnce() {
|
|
123
162
|
try {
|
|
124
163
|
const next = await fetchOnce();
|
|
125
164
|
if (next && next.instance && next.instance !== instance) {
|
|
@@ -196,8 +235,28 @@ function createLmxDiscovery(opts = {}) {
|
|
|
196
235
|
timer = null;
|
|
197
236
|
}
|
|
198
237
|
|
|
238
|
+
/**
|
|
239
|
+
* Apply new settings to a live poller (0.25.0). `undefined` keeps the current value. A new poll
|
|
240
|
+
* interval restarts a running timer so it takes effect now, not after a process restart.
|
|
241
|
+
* The status URL and instance are identity, not settings: a different URL is a different poller
|
|
242
|
+
* (the registry replaces it — see providers/lmx.js).
|
|
243
|
+
*/
|
|
244
|
+
function configure(next = {}) {
|
|
245
|
+
if (next.token !== undefined) token = next.token;
|
|
246
|
+
if (next.ca !== undefined) ca = next.ca || null;
|
|
247
|
+
if (next.staleMs !== undefined) staleMs = next.staleMs || STALE_MS;
|
|
248
|
+
if (next.fetchTimeoutMs !== undefined) fetchTimeoutMs = Number(next.fetchTimeoutMs) || FETCH_TIMEOUT_MS;
|
|
249
|
+
if (next.pollMs !== undefined) {
|
|
250
|
+
const p = floorPoll(next.pollMs);
|
|
251
|
+
if (p !== pollMs) {
|
|
252
|
+
pollMs = p;
|
|
253
|
+
if (timer) { stop(); timer = setInterval(refresh, pollMs); if (timer.unref) timer.unref(); }
|
|
254
|
+
}
|
|
255
|
+
}
|
|
256
|
+
}
|
|
257
|
+
|
|
199
258
|
return {
|
|
200
|
-
start, stop, refresh, resolve, engines, engine,
|
|
259
|
+
start, stop, refresh, resolve, engines, engine, configure,
|
|
201
260
|
/** The pinned transport, so ENGINE calls reach the same host over the same trust. */
|
|
202
261
|
transport,
|
|
203
262
|
/** For a diagnostics panel: what we know and how old it is. */
|
|
@@ -208,7 +267,9 @@ function createLmxDiscovery(opts = {}) {
|
|
|
208
267
|
engineCount: engines().length,
|
|
209
268
|
ageSec: doc ? ageSec() : null,
|
|
210
269
|
stale: doc ? isStale() : true,
|
|
211
|
-
lastError
|
|
270
|
+
lastError,
|
|
271
|
+
pollMs,
|
|
272
|
+
running: !!timer
|
|
212
273
|
})
|
|
213
274
|
};
|
|
214
275
|
}
|
|
@@ -227,4 +288,4 @@ function findEngine(list, ref) {
|
|
|
227
288
|
return all.find((e) => e && e.name === r.name) || null;
|
|
228
289
|
}
|
|
229
290
|
|
|
230
|
-
module.exports = { createLmxDiscovery, findEngine, POLL_MS, STALE_MS, HEALTHY };
|
|
291
|
+
module.exports = { createLmxDiscovery, findEngine, POLL_MS, STALE_MS, HEALTHY, MIN_POLL_MS, FETCH_TIMEOUT_MS };
|
|
@@ -154,8 +154,17 @@ async function complete(cfg, opts) {
|
|
|
154
154
|
// that thinking back separately (`reasoning_content`) or inline, wrapped in <think> tags. Neither
|
|
155
155
|
// is the answer, and both have to be recognised — otherwise a model that reasoned and then ran
|
|
156
156
|
// out of room looks identical to a broken server.
|
|
157
|
-
|
|
158
|
-
|
|
157
|
+
let reasoning = message.reasoning_content || message.reasoning || '';
|
|
158
|
+
let text = stripThinking(message.content || '');
|
|
159
|
+
|
|
160
|
+
// A LENGTH STOP INSIDE AN UNCLOSED <think> IS NO ANSWER (0.26.0). stripThinking only removes a
|
|
161
|
+
// CLOSED block, so a model that ran out of room mid-thought used to resolve with its half-finished
|
|
162
|
+
// reasoning as the "answer" (truncated: true). Narrow on purpose: only a reply that STARTS with
|
|
163
|
+
// the tag, has no closing tag, and stopped on length — the case reasoning: 'full' makes common.
|
|
164
|
+
if (finishReason === 'length' && unclosedThinking(text)) {
|
|
165
|
+
reasoning = reasoning || text;
|
|
166
|
+
text = '';
|
|
167
|
+
}
|
|
159
168
|
|
|
160
169
|
if (!text) {
|
|
161
170
|
throw emptyCompletion({ label, finishReason, reasoning, maxTokens: body.max_tokens, usage: payload.usage });
|
|
@@ -214,6 +223,12 @@ function stripThinking(content) {
|
|
|
214
223
|
.trim();
|
|
215
224
|
}
|
|
216
225
|
|
|
226
|
+
/** A reply that opens a <think>/<thinking> block and never closes it. */
|
|
227
|
+
function unclosedThinking(text) {
|
|
228
|
+
const m = /^<(think|thinking)>/i.exec(String(text || '').trimStart());
|
|
229
|
+
return !!m && !new RegExp(`</${m[1]}>`, 'i').test(text);
|
|
230
|
+
}
|
|
231
|
+
|
|
217
232
|
/**
|
|
218
233
|
* Why nothing came back — the message an operator can act on.
|
|
219
234
|
*
|
|
@@ -227,7 +242,8 @@ function emptyCompletion({ label, finishReason, reasoning, maxTokens, usage }) {
|
|
|
227
242
|
return new AiError('bad_response',
|
|
228
243
|
`${label} ran out of room before it answered — it used all ${maxTokens} reply tokens` +
|
|
229
244
|
(reasoning ? ' on internal reasoning' : '') +
|
|
230
|
-
'. This model thinks before it replies, so raise the reply limit (1024 is a sensible floor
|
|
245
|
+
'. This model thinks before it replies, so raise the reply limit (1024 is a sensible floor; ' +
|
|
246
|
+
"4096 or more with reasoning: 'full').",
|
|
231
247
|
{ usage });
|
|
232
248
|
}
|
|
233
249
|
if (reasoning) {
|