@aria-framework/ai 0.24.0 → 0.25.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +18 -11
- package/index.js +17 -24
- package/lmxStatus.js +5 -1
- package/lmxVerify.js +5 -0
- package/package.json +1 -1
- package/providers/lmx.js +48 -16
- package/providers/lmxDiscovery.js +71 -10
- package/providers/openai-compatible.js +8 -0
package/README.md
CHANGED
|
@@ -29,8 +29,7 @@ const ai = createAiClient({
|
|
|
29
29
|
timeoutMs: 60000, maxTokens: 1024, contextTokens: 8192
|
|
30
30
|
}),
|
|
31
31
|
budget: { async assertWithinBudget(cfg, meta) {}, async record(cfg, result, meta) {} }, // optional
|
|
32
|
-
logger: console
|
|
33
|
-
maxRetryAfterMs: 10000 // optional: the longest Retry-After a call will wait out (see 429 below)
|
|
32
|
+
logger: console // optional
|
|
34
33
|
});
|
|
35
34
|
|
|
36
35
|
const r = await ai.complete({ system, messages, maxTokens: 400, schema /* optional */ });
|
|
@@ -55,7 +54,7 @@ to write yourself:
|
|
|
55
54
|
|
|
56
55
|
| Contract rule | What the adapter does |
|
|
57
56
|
|---|---|
|
|
58
|
-
| Poll `/status` about every 2 s | One poller per `
|
|
57
|
+
| Poll `/status` about every 2 s | One poller per stack (keyed by `instance`), shared by every engine on it, `unref()`'d. Each read is bounded (`statusTimeoutMs`, default 5 s), concurrent reads share one fetch, and `pollMs` is floored at 500 ms |
|
|
59
58
|
| Route only to `healthy`; `draining` gets nothing new | Selects **for** `healthy`, so a state lmx adds later can never receive work by accident |
|
|
60
59
|
| Engine URLs come from the document | Resolved on every call, and never stored |
|
|
61
60
|
| A status outage is not an inference outage | Keeps routing on the last good document for 3 minutes (`staleMs`), logging loudly |
|
|
@@ -63,8 +62,8 @@ to write yourself:
|
|
|
63
62
|
| Choose the reasoning flag from the model | `qwen` → `chat_template_kwargs.enable_thinking=false`, `gpt-oss` → `reasoning_effort:'low'`, anything else gets neither (with a warning) |
|
|
64
63
|
| Size requests against `maxInputTokens` (per slot) | `contextTokens` is taken from the engine, not from config |
|
|
65
64
|
| Identity is `(instance, id)`, falling back to `name` | `lmx.engineId` is matched first; `rekeyPlan()` migrates rows when ids are minted or engines renamed |
|
|
66
|
-
| A `429` is about the key, not the engine |
|
|
67
|
-
| The document's `instance` must match | A document for another instance is refused, because engine names collide across stacks |
|
|
65
|
+
| A `429` is about the key, not the engine | Fails fast (since 0.25.0): throws `rate_limit` with `lmxSkip: 'lmx_throttled'` and `retryAfterMs`. The client never waits; whether to wait or move on is the caller's decision |
|
|
66
|
+
| The document's `instance` must match | A document for another instance is refused, because engine names collide across stacks. `lmxVerify` then returns `engines: null`, so nothing downstream (`rekeyPlan`) can act on another stack's list |
|
|
68
67
|
|
|
69
68
|
### Ask the operator for
|
|
70
69
|
|
|
@@ -94,7 +93,8 @@ The adapter takes the status URL exactly as given; it does not append `/status`.
|
|
|
94
93
|
ca: '-----BEGIN CERTIFICATE-----…', // PEM to pin; null = system trust store
|
|
95
94
|
engine: 'advanced', // engine name: always stored, used for display and logs
|
|
96
95
|
engineId: '7c9e6679-…', // engine id when the stack publishes one; null before mint-ids
|
|
97
|
-
pollMs: 2000, staleMs: 180000
|
|
96
|
+
pollMs: 2000, staleMs: 180000, // optional; pollMs is floored at 500, null/undefined = default
|
|
97
|
+
statusTimeoutMs: 5000 // optional; one status read's deadline
|
|
98
98
|
}
|
|
99
99
|
}
|
|
100
100
|
```
|
|
@@ -142,6 +142,14 @@ const { findEngine } = require('@aria-framework/ai');
|
|
|
142
142
|
const engine = findEngine(doc.engines, { id: row.lmx_engine_id || null, name: row.lmx_engine });
|
|
143
143
|
```
|
|
144
144
|
|
|
145
|
+
### Rotating credentials, and removing a stack
|
|
146
|
+
|
|
147
|
+
The poller takes its token, certificate and intervals from the config you pass on **every** call
|
|
148
|
+
(`discoveryFor` reconfigures the live poller since 0.25.0), so a rotated token or a replaced
|
|
149
|
+
certificate takes effect on the next call — no restart. A changed `statusUrl` replaces the poller.
|
|
150
|
+
When you delete or disable a supervisor, call
|
|
151
|
+
`require('@aria-framework/ai/providers/lmx').forgetDiscovery(instance)` so its poller stops.
|
|
152
|
+
|
|
145
153
|
### Errors and routing
|
|
146
154
|
|
|
147
155
|
When an lmx call cannot run, the thrown `AiError` carries `err.lmxSkip`. None of these mean the
|
|
@@ -153,7 +161,7 @@ instead:
|
|
|
153
161
|
| `lmx_not_healthy` | draining, restarting or starting; the supervisor is doing planned work |
|
|
154
162
|
| `lmx_stale` | no fresh status document, so we cannot see the stack |
|
|
155
163
|
| `lmx_unknown_engine` | the engine is not in the document (removed, or hidden from this gateway key) |
|
|
156
|
-
| `lmx_throttled` | a `429` on the key
|
|
164
|
+
| `lmx_throttled` | a `429` on the gateway key. Not retried by the client; `err.retryAfterMs` says how long the gateway asked for. Every engine on the stack shares the key, so the next endpoint on the same stack will likely be throttled too |
|
|
157
165
|
|
|
158
166
|
Failover across several engines for one job is the app's job; lmx provides none. List
|
|
159
167
|
endpoints in preference order and take the first one that serves.
|
|
@@ -161,10 +169,9 @@ endpoints in preference order and take the first one that serves.
|
|
|
161
169
|
### Embeddings
|
|
162
170
|
|
|
163
171
|
Call `client.embed(cfg, texts, { signal })` and pass your own resolved embedding config; it never
|
|
164
|
-
falls back to `resolveConfig()`. It applies the same
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
Embeddings are not metered against the token budget.
|
|
172
|
+
falls back to `resolveConfig()`. It applies the same policy as `complete()`: a fast dropped
|
|
173
|
+
connection is retried once, a `429` fails fast with `retryAfterMs` on the error, and a config with
|
|
174
|
+
`enabled: false` is refused. Embeddings are not metered against the token budget.
|
|
168
175
|
|
|
169
176
|
On lmx, `embed()` puts the engine's `dimensions` on the config as `embeddingDimensions`. Record the width
|
|
170
177
|
you built an index with and compare it on startup. A model swap changes the width, and an index
|
package/index.js
CHANGED
|
@@ -50,8 +50,6 @@ const DEFAULTS = {
|
|
|
50
50
|
|
|
51
51
|
const RETRY_AFTER_MS = 400;
|
|
52
52
|
const RETRY_ONLY_IF_FAILED_WITHIN_MS = 5000;
|
|
53
|
-
/** The longest Retry-After a call will sit out before giving up on this endpoint. */
|
|
54
|
-
const MAX_RETRY_AFTER_MS = 10000;
|
|
55
53
|
|
|
56
54
|
const NOOP_LOGGER = { info() {}, warn() {}, error() {} };
|
|
57
55
|
const NOOP_BUDGET = { async assertWithinBudget() {}, async record() {} };
|
|
@@ -67,13 +65,17 @@ function createAiClient(deps = {}) {
|
|
|
67
65
|
}
|
|
68
66
|
const log = deps.logger || NOOP_LOGGER;
|
|
69
67
|
const meter = deps.budget || NOOP_BUDGET;
|
|
70
|
-
const maxRetryAfterMs = Number.isFinite(deps.maxRetryAfterMs) ? deps.maxRetryAfterMs : MAX_RETRY_AFTER_MS;
|
|
71
68
|
|
|
72
69
|
/**
|
|
73
|
-
* Try once more, but only for the
|
|
74
|
-
* dropped the connection while loading a model,
|
|
75
|
-
*
|
|
76
|
-
*
|
|
70
|
+
* Try once more, but only for the failure where trying again could help — a local provider that
|
|
71
|
+
* dropped the connection while loading a model. A cancelled call, a timeout, a rate limit or a slow
|
|
72
|
+
* failure is never retried (see the guards below).
|
|
73
|
+
*
|
|
74
|
+
* A 429 FAILS FAST (0.25.0). 0.23/0.24 slept its Retry-After here (up to 10 s), below every app's
|
|
75
|
+
* dispatcher: every engine on an lmx stack shares one gateway key, so each failover re-hit the same
|
|
76
|
+
* throttle and paid the wait again, and an OpenAI-compatible primary that sent Retry-After delayed a
|
|
77
|
+
* healthy backup by the full wait. Decided 2026-10-01: the client does not wait. The error carries
|
|
78
|
+
* `retryAfterMs` (and `lmxSkip: 'lmx_throttled'` from an lmx engine) for a caller that chooses to.
|
|
77
79
|
*/
|
|
78
80
|
async function withOneRetry(run, opts = {}) {
|
|
79
81
|
const startedAt = Date.now();
|
|
@@ -82,17 +84,6 @@ function createAiClient(deps = {}) {
|
|
|
82
84
|
} catch (err) {
|
|
83
85
|
if (opts.signal && opts.signal.aborted) throw err;
|
|
84
86
|
if (!err || !err.retryable) throw err;
|
|
85
|
-
// A 429 THAT SAYS HOW LONG. The lmx gateway's per-key limits answer with Retry-After, and its
|
|
86
|
-
// contract is explicit: the limit is on the KEY, so wait and retry rather than blaming the
|
|
87
|
-
// engine. Bounded, because a request a person is waiting on cannot sit out a minute; past the
|
|
88
|
-
// cap it throws as before and the caller's routing decides.
|
|
89
|
-
if (err.kind === 'rate_limit' && Number.isFinite(err.retryAfterMs)
|
|
90
|
-
&& err.retryAfterMs <= maxRetryAfterMs) {
|
|
91
|
-
log.warn(`AI: rate limited — waiting ${err.retryAfterMs}ms as asked, then trying once more`);
|
|
92
|
-
await new Promise((r) => setTimeout(r, err.retryAfterMs));
|
|
93
|
-
if (opts.signal && opts.signal.aborted) throw err;
|
|
94
|
-
return run();
|
|
95
|
-
}
|
|
96
87
|
const elapsed = Date.now() - startedAt;
|
|
97
88
|
// A TIMEOUT is the deadline itself being reached — retrying waits the whole deadline again. A
|
|
98
89
|
// RATE LIMIT is the provider asking for less pressure. A slow `unreachable` is not the
|
|
@@ -146,12 +137,9 @@ function createAiClient(deps = {}) {
|
|
|
146
137
|
}
|
|
147
138
|
|
|
148
139
|
/**
|
|
149
|
-
* Embed one or more strings, with the same retry policy as complete()
|
|
150
|
-
*
|
|
151
|
-
*
|
|
152
|
-
* the adapter directly, so a gateway 429 failed the whole batch even though Retry-After said how
|
|
153
|
-
* long to wait - the README promised otherwise, and lmx's CLIENT.md requires a wait-and-retry for
|
|
154
|
-
* every request through the gateway. Embeddings are idempotent, so a resend is always safe.
|
|
140
|
+
* Embed one or more strings, with the same retry policy as complete() — a fast dropped connection
|
|
141
|
+
* is retried once; a 429 fails fast (0.25.0, see withOneRetry). Embeddings are idempotent, so the
|
|
142
|
+
* one resend is always safe.
|
|
155
143
|
*
|
|
156
144
|
* THE CONFIG IS THE CALLER'S. An app resolves its embedding settings separately from completion
|
|
157
145
|
* (embeddingModel, often a different endpoint), so this never falls back to resolveConfig().
|
|
@@ -161,6 +149,11 @@ function createAiClient(deps = {}) {
|
|
|
161
149
|
* @param {{signal?: AbortSignal}} [opts]
|
|
162
150
|
*/
|
|
163
151
|
async function embed(cfg, texts, opts = {}) {
|
|
152
|
+
// EXPLICITLY DISABLED is refused, as complete() refuses it (the review found embed ignored it).
|
|
153
|
+
// Strictly `false`: an embedding config the caller built without the flag is still honoured.
|
|
154
|
+
if (cfg && cfg.enabled === false) {
|
|
155
|
+
throw new AiError('disabled', 'This endpoint is switched off.');
|
|
156
|
+
}
|
|
164
157
|
const adapter = cfg && PROVIDERS[cfg.provider];
|
|
165
158
|
if (!adapter) {
|
|
166
159
|
throw new AiError('unconfigured', `No adapter is registered for provider "${cfg && cfg.provider}".`);
|
package/lmxStatus.js
CHANGED
|
@@ -155,12 +155,16 @@ const rowRef = (row) => ({ id: row.lmx_engine_id || null, name: row.lmx_engine }
|
|
|
155
155
|
*/
|
|
156
156
|
function rekeyPlan(rows, engines) {
|
|
157
157
|
if (!Array.isArray(engines)) return [];
|
|
158
|
+
// IDS ROWS ALREADY TRACK (0.25.0). CLIENT.md's condition for minting is "an id you have never
|
|
159
|
+
// seen". Without it, renaming row A's engine to the name a stale, id-less row B still carries
|
|
160
|
+
// handed B A's id — B's routes silently went to A's engine.
|
|
161
|
+
const held = new Set((rows || []).map((r) => r && r.lmx_engine_id).filter(Boolean));
|
|
158
162
|
const plan = [];
|
|
159
163
|
for (const row of rows || []) {
|
|
160
164
|
if (!row) continue;
|
|
161
165
|
if (!row.lmx_engine_id) {
|
|
162
166
|
const e = findEngine(engines, { name: row.lmx_engine });
|
|
163
|
-
if (e && e.id) plan.push({ id: row.id, set: { lmx_engine_id: e.id }, why: 'minted' });
|
|
167
|
+
if (e && e.id && !held.has(e.id)) plan.push({ id: row.id, set: { lmx_engine_id: e.id }, why: 'minted' });
|
|
164
168
|
continue;
|
|
165
169
|
}
|
|
166
170
|
const e = findEngine(engines, { id: row.lmx_engine_id });
|
package/lmxVerify.js
CHANGED
|
@@ -216,6 +216,11 @@ async function verify(o = {}) {
|
|
|
216
216
|
if (doc && doc.instance && String(doc.instance) !== instance) {
|
|
217
217
|
checks.push(fail('instance',
|
|
218
218
|
`this stack reports itself as “${doc.instance}”, not “${instance}” — the address points at a different deployment`));
|
|
219
|
+
// AND ITS ENGINES ARE NOT HANDED BACK (0.25.0). The apps' scheduled refresh feeds
|
|
220
|
+
// report.engines to rekeyPlan; returning another deployment's list wrote THAT stack's engine
|
|
221
|
+
// ids onto this stack's rows — irreversibly, since a row with an id is never re-pointed.
|
|
222
|
+
// `null` is "no document for this stack", which every consumer treats as "write nothing".
|
|
223
|
+
found = null;
|
|
219
224
|
}
|
|
220
225
|
|
|
221
226
|
checks.push(await enginesKeyCheck(o, engines, timeoutMs));
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@aria-framework/ai",
|
|
3
3
|
"description": "Aria App Framework \u2014 AI module. A dependency-injected model seam (createAiClient) over several providers (LM Studio / OpenAI-compatible / Anthropic), with a fact-preservation guard, generic Polish and Generate writing engines, and a browser polish widget. Prompts and config stay in the consuming app.",
|
|
4
|
-
"version": "0.
|
|
4
|
+
"version": "0.25.0",
|
|
5
5
|
"license": "UNLICENSED",
|
|
6
6
|
"private": false,
|
|
7
7
|
"publishConfig": {
|
package/providers/lmx.js
CHANGED
|
@@ -27,16 +27,33 @@ const { createLmxDiscovery } = require('./lmxDiscovery');
|
|
|
27
27
|
const { lmxTransport } = require('./lmxTransport');
|
|
28
28
|
const { AiError } = require('../error');
|
|
29
29
|
|
|
30
|
-
/**
|
|
30
|
+
/**
|
|
31
|
+
* Discovery instances, ONE PER STACK (keyed by instance), so several engines on one stack share a
|
|
32
|
+
* single poller.
|
|
33
|
+
*
|
|
34
|
+
* Keyed by instance alone since 0.25.0. Keyed by (instance, statusUrl), an edited status URL created
|
|
35
|
+
* a SECOND poller and left the first polling the old address every 2 s for the life of the process.
|
|
36
|
+
*/
|
|
31
37
|
const registry = new Map();
|
|
32
38
|
|
|
33
|
-
|
|
39
|
+
/** The settings a live poller takes from the app's supervisor row, applied on every use. */
|
|
40
|
+
const settingsOf = (lmx) => ({
|
|
41
|
+
token: lmx.statusToken,
|
|
42
|
+
ca: lmx.ca || null,
|
|
43
|
+
staleMs: lmx.staleMs,
|
|
44
|
+
pollMs: lmx.pollMs,
|
|
45
|
+
fetchTimeoutMs: lmx.statusTimeoutMs
|
|
46
|
+
});
|
|
34
47
|
|
|
35
48
|
/**
|
|
36
|
-
* The discovery for this supervisor, created once and started on first use
|
|
49
|
+
* The discovery for this supervisor, created once and started on first use — and RECONFIGURED on
|
|
50
|
+
* every use (0.25.0).
|
|
37
51
|
*
|
|
38
|
-
* Credentials are
|
|
39
|
-
*
|
|
52
|
+
* Credentials are not part of the key, so rotating a token does not orphan the poller; until 0.25.0
|
|
53
|
+
* nothing then applied the new token either (the comment here said "updated in place"; nothing did),
|
|
54
|
+
* so a rotated key or a replaced certificate never reached the poller and every lmx endpoint went
|
|
55
|
+
* stale until a restart. configure() applies whatever the app hands over each time — the app
|
|
56
|
+
* resolves its supervisor row per call, so the row is the source of truth.
|
|
40
57
|
*/
|
|
41
58
|
function discoveryFor(lmx, logger) {
|
|
42
59
|
if (!lmx || !lmx.instance || !lmx.statusUrl) {
|
|
@@ -44,29 +61,44 @@ function discoveryFor(lmx, logger) {
|
|
|
44
61
|
'This endpoint is an lmx engine but no supervisor is configured for it — an engine name '
|
|
45
62
|
+ 'without a status listener cannot be resolved to an address.');
|
|
46
63
|
}
|
|
47
|
-
const key =
|
|
64
|
+
const key = lmx.instance;
|
|
48
65
|
let d = registry.get(key);
|
|
49
|
-
if (
|
|
50
|
-
|
|
66
|
+
if (d && d.status().statusUrl !== lmx.statusUrl) {
|
|
67
|
+
// A NEW ADDRESS IS A NEW POLLER: the document, its age and its last error belong to the old one.
|
|
68
|
+
d.stop();
|
|
69
|
+
registry.delete(key);
|
|
70
|
+
d = null;
|
|
71
|
+
}
|
|
72
|
+
if (d) {
|
|
73
|
+
d.configure(settingsOf(lmx));
|
|
74
|
+
} else {
|
|
75
|
+
d = createLmxDiscovery(Object.assign({
|
|
51
76
|
instance: lmx.instance,
|
|
52
77
|
statusUrl: lmx.statusUrl,
|
|
53
|
-
token: lmx.statusToken,
|
|
54
|
-
ca: lmx.ca || null,
|
|
55
|
-
staleMs: lmx.staleMs,
|
|
56
|
-
// Was accepted from the app's supervisor row and then dropped here, so a configured poll
|
|
57
|
-
// interval never took effect. undefined falls through to POLL_MS.
|
|
58
|
-
pollMs: lmx.pollMs,
|
|
59
78
|
// Test seam, the same one createLmxDiscovery already takes: it is the only way to exercise
|
|
60
79
|
// complete() end-to-end without a live stack, and the cold-start bug lived exactly there.
|
|
61
80
|
fetchImpl: lmx.fetchImpl,
|
|
62
81
|
logger
|
|
63
|
-
});
|
|
82
|
+
}, settingsOf(lmx)));
|
|
64
83
|
d.start();
|
|
65
84
|
registry.set(key, d);
|
|
66
85
|
}
|
|
67
86
|
return d;
|
|
68
87
|
}
|
|
69
88
|
|
|
89
|
+
/**
|
|
90
|
+
* Stop and drop the poller for a stack — call it when the app deletes or disables a supervisor
|
|
91
|
+
* (0.25.0). Otherwise that stack's poller kept polling every 2 s for the life of the process.
|
|
92
|
+
* @returns {boolean} whether there was one to forget
|
|
93
|
+
*/
|
|
94
|
+
function forgetDiscovery(instance) {
|
|
95
|
+
const d = registry.get(instance);
|
|
96
|
+
if (!d) return false;
|
|
97
|
+
d.stop();
|
|
98
|
+
registry.delete(instance);
|
|
99
|
+
return true;
|
|
100
|
+
}
|
|
101
|
+
|
|
70
102
|
/**
|
|
71
103
|
* Why a resolution failed, in the vocabulary the dispatcher classifies on.
|
|
72
104
|
*
|
|
@@ -257,5 +289,5 @@ function _resetRegistry() {
|
|
|
257
289
|
module.exports = {
|
|
258
290
|
complete, embed, listModels, listModelsResult,
|
|
259
291
|
apiRoot: openai.apiRoot,
|
|
260
|
-
reasoningFor, discoveryFor, _resetRegistry, SKIP_CODE
|
|
292
|
+
reasoningFor, discoveryFor, forgetDiscovery, _resetRegistry, SKIP_CODE
|
|
261
293
|
};
|
|
@@ -56,6 +56,25 @@ const STALE_MS = 3 * 60 * 1000;
|
|
|
56
56
|
/** The only state that may receive new work. */
|
|
57
57
|
const HEALTHY = 'healthy';
|
|
58
58
|
|
|
59
|
+
/**
|
|
60
|
+
* The fastest a poller may run (0.25.0). A poll interval of 0 — a NULL coerced, a typo — is not a
|
|
61
|
+
* faster poll, it is a busy loop against the supervisor.
|
|
62
|
+
*/
|
|
63
|
+
const MIN_POLL_MS = 500;
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* How long one status read may take (0.25.0). There was no bound at all — only undici's 300 s
|
|
67
|
+
* defaults — so after a restart an lmx call awaited a hung listener far past its own timeout and
|
|
68
|
+
* the dispatcher never failed over. The document is small and served no-store: seconds is ample.
|
|
69
|
+
*/
|
|
70
|
+
const FETCH_TIMEOUT_MS = 5000;
|
|
71
|
+
|
|
72
|
+
// null/undefined mean "the default" (lmxStore stores NULL for exactly that); any number — 0, a
|
|
73
|
+
// typo, NaN — is floored rather than trusted.
|
|
74
|
+
const floorPoll = (ms) => (ms === undefined || ms === null || ms === ''
|
|
75
|
+
? POLL_MS
|
|
76
|
+
: Math.max(MIN_POLL_MS, Number(ms) || 0));
|
|
77
|
+
|
|
59
78
|
/**
|
|
60
79
|
* @param {object} opts
|
|
61
80
|
* @param {string} opts.statusUrl e.g. https://host:9443/status
|
|
@@ -63,7 +82,8 @@ const HEALTHY = 'healthy';
|
|
|
63
82
|
* @param {string} opts.token bearer for the status listener
|
|
64
83
|
* @param {string} [opts.ca] PEM of the certificate to pin
|
|
65
84
|
* @param {number} [opts.staleMs]
|
|
66
|
-
* @param {number} [opts.pollMs]
|
|
85
|
+
* @param {number} [opts.pollMs] floored at MIN_POLL_MS
|
|
86
|
+
* @param {number} [opts.fetchTimeoutMs] one status read's deadline (default FETCH_TIMEOUT_MS)
|
|
67
87
|
* @param {object} [opts.logger]
|
|
68
88
|
* @param {function} [opts.fetchImpl] injectable for tests; defaults to global fetch
|
|
69
89
|
* @param {function} [opts.now] injectable clock
|
|
@@ -72,10 +92,6 @@ function createLmxDiscovery(opts = {}) {
|
|
|
72
92
|
const {
|
|
73
93
|
statusUrl,
|
|
74
94
|
instance,
|
|
75
|
-
token,
|
|
76
|
-
ca = null,
|
|
77
|
-
staleMs = STALE_MS,
|
|
78
|
-
pollMs = POLL_MS,
|
|
79
95
|
logger = console,
|
|
80
96
|
fetchImpl,
|
|
81
97
|
now = () => Date.now()
|
|
@@ -84,10 +100,21 @@ function createLmxDiscovery(opts = {}) {
|
|
|
84
100
|
if (!statusUrl) throw new Error('createLmxDiscovery: statusUrl is required');
|
|
85
101
|
if (!instance) throw new Error('createLmxDiscovery: instance is required — identity is (instance, name)');
|
|
86
102
|
|
|
103
|
+
// MUTABLE, AND THAT IS THE 0.25.0 FIX. These were captured once for the life of the process, so a
|
|
104
|
+
// rotated token or a replaced certificate never reached the poller: Check (a fresh fetch) said
|
|
105
|
+
// "verified" while every poll got 401 and, after staleMs, every engine was skipped as stale until
|
|
106
|
+
// a restart. configure() below updates them in place; the registry calls it on every use.
|
|
107
|
+
let token = opts.token;
|
|
108
|
+
let ca = opts.ca || null;
|
|
109
|
+
let staleMs = opts.staleMs || STALE_MS;
|
|
110
|
+
let pollMs = floorPoll(opts.pollMs);
|
|
111
|
+
let fetchTimeoutMs = Number(opts.fetchTimeoutMs) || FETCH_TIMEOUT_MS;
|
|
112
|
+
|
|
87
113
|
let doc = null; // the last document that parsed and matched our instance
|
|
88
114
|
let docAt = 0;
|
|
89
115
|
let lastError = null;
|
|
90
116
|
let timer = null;
|
|
117
|
+
let inFlight = null; // the one refresh running now; concurrent callers share it
|
|
91
118
|
|
|
92
119
|
/**
|
|
93
120
|
* The pinned transport — see providers/lmxTransport.js for why this is not NODE_EXTRA_CA_CERTS
|
|
@@ -102,7 +129,9 @@ function createLmxDiscovery(opts = {}) {
|
|
|
102
129
|
const t = transport();
|
|
103
130
|
const res = await t.fetch(statusUrl, {
|
|
104
131
|
headers: { Authorization: `Bearer ${token}` },
|
|
105
|
-
dispatcher: t.dispatcher
|
|
132
|
+
dispatcher: t.dispatcher,
|
|
133
|
+
// BOUNDED (0.25.0): a listener that accepts and never answers is abandoned, not awaited.
|
|
134
|
+
signal: AbortSignal.timeout(fetchTimeoutMs)
|
|
106
135
|
});
|
|
107
136
|
if (!res.ok) {
|
|
108
137
|
const err = new Error(`status listener answered ${res.status}`);
|
|
@@ -119,7 +148,17 @@ function createLmxDiscovery(opts = {}) {
|
|
|
119
148
|
* failure — the listener answered perfectly — so it is reported as a configuration error, which
|
|
120
149
|
* is what it is. Adopting it would route `analysis` to another deployment's `analysis`.
|
|
121
150
|
*/
|
|
122
|
-
|
|
151
|
+
/**
|
|
152
|
+
* SINGLE-FLIGHT (0.25.0). The 2 s poller and every warming AI call each started their own read, so
|
|
153
|
+
* a hung listener stacked a new pending fetch per tick and per request. Concurrent callers now
|
|
154
|
+
* share the one in flight.
|
|
155
|
+
*/
|
|
156
|
+
function refresh() {
|
|
157
|
+
if (!inFlight) inFlight = refreshOnce().finally(() => { inFlight = null; });
|
|
158
|
+
return inFlight;
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
async function refreshOnce() {
|
|
123
162
|
try {
|
|
124
163
|
const next = await fetchOnce();
|
|
125
164
|
if (next && next.instance && next.instance !== instance) {
|
|
@@ -196,8 +235,28 @@ function createLmxDiscovery(opts = {}) {
|
|
|
196
235
|
timer = null;
|
|
197
236
|
}
|
|
198
237
|
|
|
238
|
+
/**
|
|
239
|
+
* Apply new settings to a live poller (0.25.0). `undefined` keeps the current value. A new poll
|
|
240
|
+
* interval restarts a running timer so it takes effect now, not after a process restart.
|
|
241
|
+
* The status URL and instance are identity, not settings: a different URL is a different poller
|
|
242
|
+
* (the registry replaces it — see providers/lmx.js).
|
|
243
|
+
*/
|
|
244
|
+
function configure(next = {}) {
|
|
245
|
+
if (next.token !== undefined) token = next.token;
|
|
246
|
+
if (next.ca !== undefined) ca = next.ca || null;
|
|
247
|
+
if (next.staleMs !== undefined) staleMs = next.staleMs || STALE_MS;
|
|
248
|
+
if (next.fetchTimeoutMs !== undefined) fetchTimeoutMs = Number(next.fetchTimeoutMs) || FETCH_TIMEOUT_MS;
|
|
249
|
+
if (next.pollMs !== undefined) {
|
|
250
|
+
const p = floorPoll(next.pollMs);
|
|
251
|
+
if (p !== pollMs) {
|
|
252
|
+
pollMs = p;
|
|
253
|
+
if (timer) { stop(); timer = setInterval(refresh, pollMs); if (timer.unref) timer.unref(); }
|
|
254
|
+
}
|
|
255
|
+
}
|
|
256
|
+
}
|
|
257
|
+
|
|
199
258
|
return {
|
|
200
|
-
start, stop, refresh, resolve, engines, engine,
|
|
259
|
+
start, stop, refresh, resolve, engines, engine, configure,
|
|
201
260
|
/** The pinned transport, so ENGINE calls reach the same host over the same trust. */
|
|
202
261
|
transport,
|
|
203
262
|
/** For a diagnostics panel: what we know and how old it is. */
|
|
@@ -208,7 +267,9 @@ function createLmxDiscovery(opts = {}) {
|
|
|
208
267
|
engineCount: engines().length,
|
|
209
268
|
ageSec: doc ? ageSec() : null,
|
|
210
269
|
stale: doc ? isStale() : true,
|
|
211
|
-
lastError
|
|
270
|
+
lastError,
|
|
271
|
+
pollMs,
|
|
272
|
+
running: !!timer
|
|
212
273
|
})
|
|
213
274
|
};
|
|
214
275
|
}
|
|
@@ -227,4 +288,4 @@ function findEngine(list, ref) {
|
|
|
227
288
|
return all.find((e) => e && e.name === r.name) || null;
|
|
228
289
|
}
|
|
229
290
|
|
|
230
|
-
module.exports = { createLmxDiscovery, findEngine, POLL_MS, STALE_MS, HEALTHY };
|
|
291
|
+
module.exports = { createLmxDiscovery, findEngine, POLL_MS, STALE_MS, HEALTHY, MIN_POLL_MS, FETCH_TIMEOUT_MS };
|
|
@@ -404,6 +404,14 @@ async function embed(cfg, texts) {
|
|
|
404
404
|
} finally {
|
|
405
405
|
clearTimeout(timer);
|
|
406
406
|
}
|
|
407
|
+
// THE STATUS FIRST, the body second — the order complete() already had. Parsing first turned a
|
|
408
|
+
// 429 (often an empty body) into "not JSON", and a 429 with a JSON body into "could not embed",
|
|
409
|
+
// both `bad_response`: never `rate_limit`, so client.embed's Retry-After wait (0.24.0) could not
|
|
410
|
+
// fire and an lmx gateway throttle read as a broken engine. httpError is handed the body already
|
|
411
|
+
// read — a real Response cannot be read twice.
|
|
412
|
+
if (res.status === 401 || res.status === 403 || res.status === 429) {
|
|
413
|
+
throw await httpError({ status: res.status, headers: res.headers, text: async () => raw }, label, cfg.apiKey);
|
|
414
|
+
}
|
|
407
415
|
let payload;
|
|
408
416
|
try { payload = JSON.parse(raw); } catch (err) {
|
|
409
417
|
throw new AiError('bad_response', `${label} returned something that is not JSON from ${url}.`);
|